K2triinK commited on
Commit
a662fa5
·
verified ·
1 Parent(s): ce6a904

Add files using upload-large-folder tool

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/README.md +58 -0
  2. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/README.md +58 -0
  3. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/README.md +209 -0
  4. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/adapter_config.json +46 -0
  5. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/chat_template.jinja +154 -0
  6. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/tokenizer_config.json +31 -0
  7. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/trainer_state.json +139 -0
  8. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/README.md +209 -0
  9. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/adapter_config.json +46 -0
  10. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/chat_template.jinja +154 -0
  11. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/tokenizer_config.json +31 -0
  12. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/trainer_state.json +1084 -0
  13. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/README.md +209 -0
  14. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/adapter_config.json +46 -0
  15. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/chat_template.jinja +154 -0
  16. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/tokenizer_config.json +31 -0
  17. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/trainer_state.json +1105 -0
  18. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/README.md +209 -0
  19. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/adapter_config.json +46 -0
  20. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/chat_template.jinja +154 -0
  21. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/tokenizer_config.json +31 -0
  22. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/trainer_state.json +1126 -0
  23. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/README.md +209 -0
  24. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/adapter_config.json +46 -0
  25. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/chat_template.jinja +154 -0
  26. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/tokenizer_config.json +31 -0
  27. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/trainer_state.json +1147 -0
  28. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/README.md +209 -0
  29. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/adapter_config.json +46 -0
  30. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/chat_template.jinja +154 -0
  31. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/tokenizer_config.json +31 -0
  32. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/trainer_state.json +1168 -0
  33. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/README.md +209 -0
  34. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/adapter_config.json +46 -0
  35. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/chat_template.jinja +154 -0
  36. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/tokenizer_config.json +31 -0
  37. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/trainer_state.json +1189 -0
  38. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1120/README.md +209 -0
  39. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1120/adapter_config.json +46 -0
  40. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1120/chat_template.jinja +154 -0
  41. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1120/tokenizer_config.json +31 -0
  42. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1120/trainer_state.json +1210 -0
  43. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1140/README.md +209 -0
  44. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1140/adapter_config.json +46 -0
  45. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1140/chat_template.jinja +154 -0
  46. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1140/tokenizer_config.json +31 -0
  47. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1140/trainer_state.json +1231 -0
  48. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1160/README.md +209 -0
  49. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1160/adapter_config.json +46 -0
  50. overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1160/chat_template.jinja +154 -0
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/README.md ADDED
@@ -0,0 +1,58 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: transformers
4
+ model_name: Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1
5
+ tags:
6
+ - generated_from_trainer
7
+ - trl
8
+ - sft
9
+ licence: license
10
+ ---
11
+
12
+ # Model Card for Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1
13
+
14
+ This model is a fine-tuned version of [Qwen/Qwen3.5-4B-Base](https://huggingface.co/Qwen/Qwen3.5-4B-Base).
15
+ It has been trained using [TRL](https://github.com/huggingface/trl).
16
+
17
+ ## Quick start
18
+
19
+ ```python
20
+ from transformers import pipeline
21
+
22
+ question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?"
23
+ generator = pipeline("text-generation", model="None", device="cuda")
24
+ output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0]
25
+ print(output["generated_text"])
26
+ ```
27
+
28
+ ## Training procedure
29
+
30
+ [<img src="https://raw.githubusercontent.com/wandb/assets/main/wandb-github-badge-28.svg" alt="Visualize in Weights & Biases" width="150" height="24"/>](https://wandb.ai/katriin-kukk/Cross_lingual_morphological_generalization/runs/mq7in2f1)
31
+
32
+
33
+
34
+ This model was trained with SFT.
35
+
36
+ ### Framework versions
37
+
38
+ - TRL: 0.29.0
39
+ - Transformers: 5.5.4
40
+ - Pytorch: 2.10.0
41
+ - Datasets: 4.6.1
42
+ - Tokenizers: 0.22.2
43
+
44
+ ## Citations
45
+
46
+
47
+
48
+ Cite TRL as:
49
+
50
+ ```bibtex
51
+ @software{vonwerra2020trl,
52
+ title = {{TRL: Transformers Reinforcement Learning}},
53
+ author = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin},
54
+ license = {Apache-2.0},
55
+ url = {https://github.com/huggingface/trl},
56
+ year = {2020}
57
+ }
58
+ ```
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/README.md ADDED
@@ -0,0 +1,58 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: transformers
4
+ model_name: Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2
5
+ tags:
6
+ - generated_from_trainer
7
+ - trl
8
+ - sft
9
+ licence: license
10
+ ---
11
+
12
+ # Model Card for Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2
13
+
14
+ This model is a fine-tuned version of [Qwen/Qwen3.5-4B-Base](https://huggingface.co/Qwen/Qwen3.5-4B-Base).
15
+ It has been trained using [TRL](https://github.com/huggingface/trl).
16
+
17
+ ## Quick start
18
+
19
+ ```python
20
+ from transformers import pipeline
21
+
22
+ question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?"
23
+ generator = pipeline("text-generation", model="None", device="cuda")
24
+ output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0]
25
+ print(output["generated_text"])
26
+ ```
27
+
28
+ ## Training procedure
29
+
30
+ [<img src="https://raw.githubusercontent.com/wandb/assets/main/wandb-github-badge-28.svg" alt="Visualize in Weights & Biases" width="150" height="24"/>](https://wandb.ai/katriin-kukk/Cross_lingual_morphological_generalization/runs/gtp4i6ta)
31
+
32
+
33
+
34
+ This model was trained with SFT.
35
+
36
+ ### Framework versions
37
+
38
+ - TRL: 0.29.0
39
+ - Transformers: 5.5.4
40
+ - Pytorch: 2.10.0
41
+ - Datasets: 4.6.1
42
+ - Tokenizers: 0.22.2
43
+
44
+ ## Citations
45
+
46
+
47
+
48
+ Cite TRL as:
49
+
50
+ ```bibtex
51
+ @software{vonwerra2020trl,
52
+ title = {{TRL: Transformers Reinforcement Learning}},
53
+ author = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin},
54
+ license = {Apache-2.0},
55
+ url = {https://github.com/huggingface/trl},
56
+ year = {2020}
57
+ }
58
+ ```
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.03828026524568501,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "v_proj",
34
+ "up_proj",
35
+ "down_proj",
36
+ "o_proj",
37
+ "q_proj",
38
+ "k_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/trainer_state.json ADDED
@@ -0,0 +1,139 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.24140012070006034,
6
+ "eval_steps": 20,
7
+ "global_step": 100,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 2.172999547421932,
14
+ "epoch": 0.04828002414001207,
15
+ "grad_norm": 2.5071208477020264,
16
+ "learning_rate": 1.7227585487708594e-05,
17
+ "loss": 1.9347333908081055,
18
+ "mean_token_accuracy": 0.6086816020309925,
19
+ "num_tokens": 47778.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.04828002414001207,
24
+ "eval_entropy": 1.6909754791956269,
25
+ "eval_loss": 1.4728385210037231,
26
+ "eval_mean_token_accuracy": 0.6773593797442619,
27
+ "eval_num_tokens": 47778.0,
28
+ "eval_runtime": 90.1265,
29
+ "eval_samples_per_second": 15.756,
30
+ "eval_steps_per_second": 1.975,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.2014626950025558,
35
+ "epoch": 0.09656004828002414,
36
+ "grad_norm": 1.5074543952941895,
37
+ "learning_rate": 3.5361886001086065e-05,
38
+ "loss": 1.0867146492004394,
39
+ "mean_token_accuracy": 0.7285927519202232,
40
+ "num_tokens": 95981.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.09656004828002414,
45
+ "eval_entropy": 1.0304168352250302,
46
+ "eval_loss": 0.9211422204971313,
47
+ "eval_mean_token_accuracy": 0.7613122848312507,
48
+ "eval_num_tokens": 95981.0,
49
+ "eval_runtime": 88.6338,
50
+ "eval_samples_per_second": 16.021,
51
+ "eval_steps_per_second": 2.008,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.9454198732972146,
56
+ "epoch": 0.14484007242003621,
57
+ "grad_norm": 1.7578108310699463,
58
+ "learning_rate": 5.349618651446354e-05,
59
+ "loss": 0.8447297096252442,
60
+ "mean_token_accuracy": 0.7723424732685089,
61
+ "num_tokens": 142784.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.14484007242003621,
66
+ "eval_entropy": 0.8647834474451086,
67
+ "eval_loss": 0.8240737318992615,
68
+ "eval_mean_token_accuracy": 0.7821227014064789,
69
+ "eval_num_tokens": 142784.0,
70
+ "eval_runtime": 88.0135,
71
+ "eval_samples_per_second": 16.134,
72
+ "eval_steps_per_second": 2.022,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.8963550612330436,
77
+ "epoch": 0.19312009656004828,
78
+ "grad_norm": 1.1464931964874268,
79
+ "learning_rate": 7.1630487027841e-05,
80
+ "loss": 0.7979836463928223,
81
+ "mean_token_accuracy": 0.7822489373385906,
82
+ "num_tokens": 187338.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.19312009656004828,
87
+ "eval_entropy": 0.8663296133614657,
88
+ "eval_loss": 0.7752988338470459,
89
+ "eval_mean_token_accuracy": 0.7885957012015782,
90
+ "eval_num_tokens": 187338.0,
91
+ "eval_runtime": 88.284,
92
+ "eval_samples_per_second": 16.084,
93
+ "eval_steps_per_second": 2.016,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.8265103206038475,
98
+ "epoch": 0.24140012070006034,
99
+ "grad_norm": 1.0106662511825562,
100
+ "learning_rate": 8.976478754121848e-05,
101
+ "loss": 0.7297060012817382,
102
+ "mean_token_accuracy": 0.7955138415098191,
103
+ "num_tokens": 236038.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24140012070006034,
108
+ "eval_entropy": 0.8122169536151244,
109
+ "eval_loss": 0.754906177520752,
110
+ "eval_mean_token_accuracy": 0.7917155255092664,
111
+ "eval_num_tokens": 236038.0,
112
+ "eval_runtime": 87.5754,
113
+ "eval_samples_per_second": 16.215,
114
+ "eval_steps_per_second": 2.033,
115
+ "step": 100
116
+ }
117
+ ],
118
+ "logging_steps": 20,
119
+ "max_steps": 4150,
120
+ "num_input_tokens_seen": 0,
121
+ "num_train_epochs": 10,
122
+ "save_steps": 20,
123
+ "stateful_callbacks": {
124
+ "TrainerControl": {
125
+ "args": {
126
+ "should_epoch_stop": false,
127
+ "should_evaluate": false,
128
+ "should_log": false,
129
+ "should_save": true,
130
+ "should_training_stop": false
131
+ },
132
+ "attributes": {}
133
+ }
134
+ },
135
+ "total_flos": 9897324005793792.0,
136
+ "train_batch_size": 4,
137
+ "trial_name": null,
138
+ "trial_params": null
139
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.03828026524568501,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "v_proj",
34
+ "up_proj",
35
+ "down_proj",
36
+ "o_proj",
37
+ "q_proj",
38
+ "k_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/trainer_state.json ADDED
@@ -0,0 +1,1084 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.4103802051901027,
6
+ "eval_steps": 20,
7
+ "global_step": 1000,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 2.172999547421932,
14
+ "epoch": 0.04828002414001207,
15
+ "grad_norm": 2.5071208477020264,
16
+ "learning_rate": 1.7227585487708594e-05,
17
+ "loss": 1.9347333908081055,
18
+ "mean_token_accuracy": 0.6086816020309925,
19
+ "num_tokens": 47778.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.04828002414001207,
24
+ "eval_entropy": 1.6909754791956269,
25
+ "eval_loss": 1.4728385210037231,
26
+ "eval_mean_token_accuracy": 0.6773593797442619,
27
+ "eval_num_tokens": 47778.0,
28
+ "eval_runtime": 90.1265,
29
+ "eval_samples_per_second": 15.756,
30
+ "eval_steps_per_second": 1.975,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.2014626950025558,
35
+ "epoch": 0.09656004828002414,
36
+ "grad_norm": 1.5074543952941895,
37
+ "learning_rate": 3.5361886001086065e-05,
38
+ "loss": 1.0867146492004394,
39
+ "mean_token_accuracy": 0.7285927519202232,
40
+ "num_tokens": 95981.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.09656004828002414,
45
+ "eval_entropy": 1.0304168352250302,
46
+ "eval_loss": 0.9211422204971313,
47
+ "eval_mean_token_accuracy": 0.7613122848312507,
48
+ "eval_num_tokens": 95981.0,
49
+ "eval_runtime": 88.6338,
50
+ "eval_samples_per_second": 16.021,
51
+ "eval_steps_per_second": 2.008,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.9454198732972146,
56
+ "epoch": 0.14484007242003621,
57
+ "grad_norm": 1.7578108310699463,
58
+ "learning_rate": 5.349618651446354e-05,
59
+ "loss": 0.8447297096252442,
60
+ "mean_token_accuracy": 0.7723424732685089,
61
+ "num_tokens": 142784.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.14484007242003621,
66
+ "eval_entropy": 0.8647834474451086,
67
+ "eval_loss": 0.8240737318992615,
68
+ "eval_mean_token_accuracy": 0.7821227014064789,
69
+ "eval_num_tokens": 142784.0,
70
+ "eval_runtime": 88.0135,
71
+ "eval_samples_per_second": 16.134,
72
+ "eval_steps_per_second": 2.022,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.8963550612330436,
77
+ "epoch": 0.19312009656004828,
78
+ "grad_norm": 1.1464931964874268,
79
+ "learning_rate": 7.1630487027841e-05,
80
+ "loss": 0.7979836463928223,
81
+ "mean_token_accuracy": 0.7822489373385906,
82
+ "num_tokens": 187338.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.19312009656004828,
87
+ "eval_entropy": 0.8663296133614657,
88
+ "eval_loss": 0.7752988338470459,
89
+ "eval_mean_token_accuracy": 0.7885957012015782,
90
+ "eval_num_tokens": 187338.0,
91
+ "eval_runtime": 88.284,
92
+ "eval_samples_per_second": 16.084,
93
+ "eval_steps_per_second": 2.016,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.8265103206038475,
98
+ "epoch": 0.24140012070006034,
99
+ "grad_norm": 1.0106662511825562,
100
+ "learning_rate": 8.976478754121848e-05,
101
+ "loss": 0.7297060012817382,
102
+ "mean_token_accuracy": 0.7955138415098191,
103
+ "num_tokens": 236038.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24140012070006034,
108
+ "eval_entropy": 0.8122169536151244,
109
+ "eval_loss": 0.754906177520752,
110
+ "eval_mean_token_accuracy": 0.7917155255092664,
111
+ "eval_num_tokens": 236038.0,
112
+ "eval_runtime": 87.5754,
113
+ "eval_samples_per_second": 16.215,
114
+ "eval_steps_per_second": 2.033,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.8098392844200134,
119
+ "epoch": 0.28968014484007243,
120
+ "grad_norm": 1.2238041162490845,
121
+ "learning_rate": 0.00010789908805459595,
122
+ "loss": 0.7201489448547364,
123
+ "mean_token_accuracy": 0.7975016921758652,
124
+ "num_tokens": 282918.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.28968014484007243,
129
+ "eval_entropy": 0.8418726462326692,
130
+ "eval_loss": 0.7393178939819336,
131
+ "eval_mean_token_accuracy": 0.7936771585700217,
132
+ "eval_num_tokens": 282918.0,
133
+ "eval_runtime": 87.1597,
134
+ "eval_samples_per_second": 16.292,
135
+ "eval_steps_per_second": 2.042,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.8080591082572937,
140
+ "epoch": 0.33796016898008446,
141
+ "grad_norm": 1.011591911315918,
142
+ "learning_rate": 0.00012603338856797343,
143
+ "loss": 0.7147697925567627,
144
+ "mean_token_accuracy": 0.7947175532579422,
145
+ "num_tokens": 331305.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.33796016898008446,
150
+ "eval_entropy": 0.7894941431083037,
151
+ "eval_loss": 0.720219075679779,
152
+ "eval_mean_token_accuracy": 0.801899236239744,
153
+ "eval_num_tokens": 331305.0,
154
+ "eval_runtime": 87.1493,
155
+ "eval_samples_per_second": 16.294,
156
+ "eval_steps_per_second": 2.042,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.7957205109298229,
161
+ "epoch": 0.38624019312009655,
162
+ "grad_norm": 0.8481830954551697,
163
+ "learning_rate": 0.00014416768908135088,
164
+ "loss": 0.7050027847290039,
165
+ "mean_token_accuracy": 0.7995368830859662,
166
+ "num_tokens": 375876.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.38624019312009655,
171
+ "eval_entropy": 0.7746140591883928,
172
+ "eval_loss": 0.7175475358963013,
173
+ "eval_mean_token_accuracy": 0.7994848955213354,
174
+ "eval_num_tokens": 375876.0,
175
+ "eval_runtime": 86.655,
176
+ "eval_samples_per_second": 16.387,
177
+ "eval_steps_per_second": 2.054,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.779565304517746,
182
+ "epoch": 0.43452021726010864,
183
+ "grad_norm": 1.032624363899231,
184
+ "learning_rate": 0.00016230198959472835,
185
+ "loss": 0.6978332042694092,
186
+ "mean_token_accuracy": 0.8015618294477462,
187
+ "num_tokens": 423913.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.43452021726010864,
192
+ "eval_entropy": 0.7455583871080634,
193
+ "eval_loss": 0.7118850350379944,
194
+ "eval_mean_token_accuracy": 0.7999682195400923,
195
+ "eval_num_tokens": 423913.0,
196
+ "eval_runtime": 86.7842,
197
+ "eval_samples_per_second": 16.362,
198
+ "eval_steps_per_second": 2.051,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.7810172818601131,
203
+ "epoch": 0.4828002414001207,
204
+ "grad_norm": 0.9276663064956665,
205
+ "learning_rate": 0.00018043629010810582,
206
+ "loss": 0.6937861442565918,
207
+ "mean_token_accuracy": 0.802893053740263,
208
+ "num_tokens": 472395.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.4828002414001207,
213
+ "eval_entropy": 0.7650193290764027,
214
+ "eval_loss": 0.7020567059516907,
215
+ "eval_mean_token_accuracy": 0.8045302153973097,
216
+ "eval_num_tokens": 472395.0,
217
+ "eval_runtime": 88.326,
218
+ "eval_samples_per_second": 16.077,
219
+ "eval_steps_per_second": 2.015,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.7862060204148292,
224
+ "epoch": 0.5310802655401328,
225
+ "grad_norm": 0.8195295929908752,
226
+ "learning_rate": 0.0001985705906214833,
227
+ "loss": 0.6878782272338867,
228
+ "mean_token_accuracy": 0.8027168907225132,
229
+ "num_tokens": 517049.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.5310802655401328,
234
+ "eval_entropy": 0.7656565583154057,
235
+ "eval_loss": 0.6973932981491089,
236
+ "eval_mean_token_accuracy": 0.8047967278555538,
237
+ "eval_num_tokens": 517049.0,
238
+ "eval_runtime": 87.7698,
239
+ "eval_samples_per_second": 16.179,
240
+ "eval_steps_per_second": 2.028,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.7611534461379051,
245
+ "epoch": 0.5793602896801449,
246
+ "grad_norm": 0.7499712109565735,
247
+ "learning_rate": 0.00021670489113486077,
248
+ "loss": 0.681937837600708,
249
+ "mean_token_accuracy": 0.8041978515684605,
250
+ "num_tokens": 562836.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.5793602896801449,
255
+ "eval_entropy": 0.7599442380197933,
256
+ "eval_loss": 0.7002046704292297,
257
+ "eval_mean_token_accuracy": 0.8051861517884759,
258
+ "eval_num_tokens": 562836.0,
259
+ "eval_runtime": 87.9239,
260
+ "eval_samples_per_second": 16.15,
261
+ "eval_steps_per_second": 2.024,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.7800783462822437,
266
+ "epoch": 0.627640313820157,
267
+ "grad_norm": 0.8645269274711609,
268
+ "learning_rate": 0.00023483919164823825,
269
+ "loss": 0.6954500675201416,
270
+ "mean_token_accuracy": 0.8031484372913837,
271
+ "num_tokens": 606612.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.627640313820157,
276
+ "eval_entropy": 0.7498679090751691,
277
+ "eval_loss": 0.7046149373054504,
278
+ "eval_mean_token_accuracy": 0.802388215667746,
279
+ "eval_num_tokens": 606612.0,
280
+ "eval_runtime": 88.2468,
281
+ "eval_samples_per_second": 16.091,
282
+ "eval_steps_per_second": 2.017,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.7586533665657044,
287
+ "epoch": 0.6759203379601689,
288
+ "grad_norm": 0.741596519947052,
289
+ "learning_rate": 0.0002529734921616157,
290
+ "loss": 0.690476369857788,
291
+ "mean_token_accuracy": 0.8046923100948333,
292
+ "num_tokens": 655920.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6759203379601689,
297
+ "eval_entropy": 0.7714274066887544,
298
+ "eval_loss": 0.6992160677909851,
299
+ "eval_mean_token_accuracy": 0.8042646482419432,
300
+ "eval_num_tokens": 655920.0,
301
+ "eval_runtime": 88.1754,
302
+ "eval_samples_per_second": 16.104,
303
+ "eval_steps_per_second": 2.019,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.7585848286747933,
308
+ "epoch": 0.724200362100181,
309
+ "grad_norm": 1.0778124332427979,
310
+ "learning_rate": 0.00027110779267499317,
311
+ "loss": 0.6790409564971924,
312
+ "mean_token_accuracy": 0.8060351841151714,
313
+ "num_tokens": 702638.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.724200362100181,
318
+ "eval_entropy": 0.7380402811457601,
319
+ "eval_loss": 0.7054820656776428,
320
+ "eval_mean_token_accuracy": 0.8003534738267406,
321
+ "eval_num_tokens": 702638.0,
322
+ "eval_runtime": 88.238,
323
+ "eval_samples_per_second": 16.093,
324
+ "eval_steps_per_second": 2.017,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.7696981988847256,
329
+ "epoch": 0.7724803862401931,
330
+ "grad_norm": 0.8408658504486084,
331
+ "learning_rate": 0.00028924209318837064,
332
+ "loss": 0.7070387363433838,
333
+ "mean_token_accuracy": 0.8035947173833847,
334
+ "num_tokens": 749377.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.7724803862401931,
339
+ "eval_entropy": 0.7897409023193831,
340
+ "eval_loss": 0.7015668749809265,
341
+ "eval_mean_token_accuracy": 0.805306687113944,
342
+ "eval_num_tokens": 749377.0,
343
+ "eval_runtime": 87.9395,
344
+ "eval_samples_per_second": 16.147,
345
+ "eval_steps_per_second": 2.024,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.762337189912796,
350
+ "epoch": 0.8207604103802052,
351
+ "grad_norm": 0.9843310117721558,
352
+ "learning_rate": 0.0003073763937017481,
353
+ "loss": 0.7021088123321533,
354
+ "mean_token_accuracy": 0.8029616877436638,
355
+ "num_tokens": 798874.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8207604103802052,
360
+ "eval_entropy": 0.7210077121016685,
361
+ "eval_loss": 0.7082746624946594,
362
+ "eval_mean_token_accuracy": 0.8049283824609906,
363
+ "eval_num_tokens": 798874.0,
364
+ "eval_runtime": 88.1706,
365
+ "eval_samples_per_second": 16.105,
366
+ "eval_steps_per_second": 2.019,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.7879025422036647,
371
+ "epoch": 0.8690404345202173,
372
+ "grad_norm": 0.9713814854621887,
373
+ "learning_rate": 0.0003255106942151256,
374
+ "loss": 0.714624547958374,
375
+ "mean_token_accuracy": 0.79987031519413,
376
+ "num_tokens": 840825.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8690404345202173,
381
+ "eval_entropy": 0.7866930489459735,
382
+ "eval_loss": 0.7145519852638245,
383
+ "eval_mean_token_accuracy": 0.8015657988157165,
384
+ "eval_num_tokens": 840825.0,
385
+ "eval_runtime": 88.3746,
386
+ "eval_samples_per_second": 16.068,
387
+ "eval_steps_per_second": 2.014,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.7844726584851742,
392
+ "epoch": 0.9173204586602294,
393
+ "grad_norm": 1.2028311491012573,
394
+ "learning_rate": 0.00034364499472850306,
395
+ "loss": 0.7040786266326904,
396
+ "mean_token_accuracy": 0.7981885217130185,
397
+ "num_tokens": 882885.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9173204586602294,
402
+ "eval_entropy": 0.7767668339643585,
403
+ "eval_loss": 0.7073574066162109,
404
+ "eval_mean_token_accuracy": 0.8040851833445303,
405
+ "eval_num_tokens": 882885.0,
406
+ "eval_runtime": 88.3828,
407
+ "eval_samples_per_second": 16.066,
408
+ "eval_steps_per_second": 2.014,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.7699790760874748,
413
+ "epoch": 0.9656004828002414,
414
+ "grad_norm": 0.9397398233413696,
415
+ "learning_rate": 0.00036177929524188054,
416
+ "loss": 0.7040313720703125,
417
+ "mean_token_accuracy": 0.8034727744758129,
418
+ "num_tokens": 927456.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9656004828002414,
423
+ "eval_entropy": 0.7749103545472863,
424
+ "eval_loss": 0.7093513011932373,
425
+ "eval_mean_token_accuracy": 0.8019482901926791,
426
+ "eval_num_tokens": 927456.0,
427
+ "eval_runtime": 86.2112,
428
+ "eval_samples_per_second": 16.471,
429
+ "eval_steps_per_second": 2.065,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.7672389172888422,
434
+ "epoch": 1.012070006035003,
435
+ "grad_norm": 0.99210524559021,
436
+ "learning_rate": 0.00037628567078152296,
437
+ "loss": 0.6913118362426758,
438
+ "mean_token_accuracy": 0.8042210944287189,
439
+ "num_tokens": 972662.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.012070006035003,
444
+ "eval_entropy": 0.7294771229283193,
445
+ "eval_loss": 0.7171520590782166,
446
+ "eval_mean_token_accuracy": 0.8015128585059991,
447
+ "eval_num_tokens": 972662.0,
448
+ "eval_runtime": 88.037,
449
+ "eval_samples_per_second": 16.13,
450
+ "eval_steps_per_second": 2.022,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.7143943183124065,
455
+ "epoch": 1.060350030175015,
456
+ "grad_norm": 1.087638258934021,
457
+ "learning_rate": 0.0003762484015601069,
458
+ "loss": 0.642704439163208,
459
+ "mean_token_accuracy": 0.8124966286122799,
460
+ "num_tokens": 1022525.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.060350030175015,
465
+ "eval_entropy": 0.743511886744017,
466
+ "eval_loss": 0.7116662859916687,
467
+ "eval_mean_token_accuracy": 0.8037210672758939,
468
+ "eval_num_tokens": 1022525.0,
469
+ "eval_runtime": 87.9931,
470
+ "eval_samples_per_second": 16.138,
471
+ "eval_steps_per_second": 2.023,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.7237346574664116,
476
+ "epoch": 1.1086300543150271,
477
+ "grad_norm": 0.9365553259849548,
478
+ "learning_rate": 0.0003761579008392123,
479
+ "loss": 0.652644681930542,
480
+ "mean_token_accuracy": 0.8107496216893196,
481
+ "num_tokens": 1069194.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1086300543150271,
486
+ "eval_entropy": 0.7372823839776972,
487
+ "eval_loss": 0.7238303422927856,
488
+ "eval_mean_token_accuracy": 0.8026230124275336,
489
+ "eval_num_tokens": 1069194.0,
490
+ "eval_runtime": 87.6778,
491
+ "eval_samples_per_second": 16.196,
492
+ "eval_steps_per_second": 2.03,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.7202137842774391,
497
+ "epoch": 1.1569100784550392,
498
+ "grad_norm": 0.7998443841934204,
499
+ "learning_rate": 0.0003760141942294726,
500
+ "loss": 0.6485219478607178,
501
+ "mean_token_accuracy": 0.8092714451253414,
502
+ "num_tokens": 1117570.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1569100784550392,
507
+ "eval_entropy": 0.71547647942318,
508
+ "eval_loss": 0.7124277949333191,
509
+ "eval_mean_token_accuracy": 0.8013174219077892,
510
+ "eval_num_tokens": 1117570.0,
511
+ "eval_runtime": 87.4387,
512
+ "eval_samples_per_second": 16.24,
513
+ "eval_steps_per_second": 2.036,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.7270819500088692,
518
+ "epoch": 1.2051901025950513,
519
+ "grad_norm": 0.9470787644386292,
520
+ "learning_rate": 0.00037581732239815854,
521
+ "loss": 0.66352219581604,
522
+ "mean_token_accuracy": 0.8102593503892421,
523
+ "num_tokens": 1161717.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.2051901025950513,
528
+ "eval_entropy": 0.7791565897759427,
529
+ "eval_loss": 0.7179726958274841,
530
+ "eval_mean_token_accuracy": 0.7948898328154275,
531
+ "eval_num_tokens": 1161717.0,
532
+ "eval_runtime": 87.9927,
533
+ "eval_samples_per_second": 16.138,
534
+ "eval_steps_per_second": 2.023,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.7320936664938926,
539
+ "epoch": 1.2534701267350634,
540
+ "grad_norm": 1.0220379829406738,
541
+ "learning_rate": 0.0003755673410576695,
542
+ "loss": 0.6581549644470215,
543
+ "mean_token_accuracy": 0.8098709337413311,
544
+ "num_tokens": 1209390.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.2534701267350634,
549
+ "eval_entropy": 0.7710678225822662,
550
+ "eval_loss": 0.708797812461853,
551
+ "eval_mean_token_accuracy": 0.8039572778042783,
552
+ "eval_num_tokens": 1209390.0,
553
+ "eval_runtime": 88.1193,
554
+ "eval_samples_per_second": 16.115,
555
+ "eval_steps_per_second": 2.02,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.7281674664467573,
560
+ "epoch": 1.3017501508750755,
561
+ "grad_norm": 0.8775188326835632,
562
+ "learning_rate": 0.000375264320949768,
563
+ "loss": 0.6631278991699219,
564
+ "mean_token_accuracy": 0.8101261422038079,
565
+ "num_tokens": 1256611.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3017501508750755,
570
+ "eval_entropy": 0.7199799800187014,
571
+ "eval_loss": 0.6981531381607056,
572
+ "eval_mean_token_accuracy": 0.8071265254127845,
573
+ "eval_num_tokens": 1256611.0,
574
+ "eval_runtime": 88.165,
575
+ "eval_samples_per_second": 16.106,
576
+ "eval_steps_per_second": 2.019,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.722860773652792,
581
+ "epoch": 1.3500301750150876,
582
+ "grad_norm": 0.8949288129806519,
583
+ "learning_rate": 0.00037490834782556,
584
+ "loss": 0.6639598846435547,
585
+ "mean_token_accuracy": 0.8105649061501026,
586
+ "num_tokens": 1302325.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3500301750150876,
591
+ "eval_entropy": 0.7134760032209118,
592
+ "eval_loss": 0.7029948234558105,
593
+ "eval_mean_token_accuracy": 0.8062985880991046,
594
+ "eval_num_tokens": 1302325.0,
595
+ "eval_runtime": 88.1286,
596
+ "eval_samples_per_second": 16.113,
597
+ "eval_steps_per_second": 2.02,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.7299126699566841,
602
+ "epoch": 1.3983101991550995,
603
+ "grad_norm": 1.0323039293289185,
604
+ "learning_rate": 0.0003744995224212291,
605
+ "loss": 0.6535763263702392,
606
+ "mean_token_accuracy": 0.8120224848389626,
607
+ "num_tokens": 1343844.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.3983101991550995,
612
+ "eval_entropy": 0.7169701182440426,
613
+ "eval_loss": 0.7067857980728149,
614
+ "eval_mean_token_accuracy": 0.8040432936689826,
615
+ "eval_num_tokens": 1343844.0,
616
+ "eval_runtime": 88.2252,
617
+ "eval_samples_per_second": 16.095,
618
+ "eval_steps_per_second": 2.018,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.7281742256134749,
623
+ "epoch": 1.4465902232951118,
624
+ "grad_norm": 1.3370424509048462,
625
+ "learning_rate": 0.00037403796042952863,
626
+ "loss": 0.6637272357940673,
627
+ "mean_token_accuracy": 0.8118261776864528,
628
+ "num_tokens": 1389471.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4465902232951118,
633
+ "eval_entropy": 0.7378843212395572,
634
+ "eval_loss": 0.7038390040397644,
635
+ "eval_mean_token_accuracy": 0.8059229371922739,
636
+ "eval_num_tokens": 1389471.0,
637
+ "eval_runtime": 87.9301,
638
+ "eval_samples_per_second": 16.149,
639
+ "eval_steps_per_second": 2.024,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.7388614989817143,
644
+ "epoch": 1.4948702474351236,
645
+ "grad_norm": 0.8464171290397644,
646
+ "learning_rate": 0.00037352379246704257,
647
+ "loss": 0.6604740142822265,
648
+ "mean_token_accuracy": 0.8102974124252796,
649
+ "num_tokens": 1435975.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.4948702474351236,
654
+ "eval_entropy": 0.7126847046814607,
655
+ "eval_loss": 0.6950703263282776,
656
+ "eval_mean_token_accuracy": 0.8085885392815879,
657
+ "eval_num_tokens": 1435975.0,
658
+ "eval_runtime": 88.2732,
659
+ "eval_samples_per_second": 16.086,
660
+ "eval_steps_per_second": 2.016,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.7258916139602661,
665
+ "epoch": 1.5431502715751357,
666
+ "grad_norm": 1.1780085563659668,
667
+ "learning_rate": 0.0003729571640372223,
668
+ "loss": 0.6529891014099121,
669
+ "mean_token_accuracy": 0.8085981778800487,
670
+ "num_tokens": 1482035.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.5431502715751357,
675
+ "eval_entropy": 0.7096509638797032,
676
+ "eval_loss": 0.690051257610321,
677
+ "eval_mean_token_accuracy": 0.8100520241796301,
678
+ "eval_num_tokens": 1482035.0,
679
+ "eval_runtime": 88.3153,
680
+ "eval_samples_per_second": 16.079,
681
+ "eval_steps_per_second": 2.016,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.7151173129677773,
686
+ "epoch": 1.5914302957151478,
687
+ "grad_norm": 1.0357667207717896,
688
+ "learning_rate": 0.0003723382354892108,
689
+ "loss": 0.6451474189758301,
690
+ "mean_token_accuracy": 0.8136902332305909,
691
+ "num_tokens": 1528119.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.5914302957151478,
696
+ "eval_entropy": 0.7060252652409371,
697
+ "eval_loss": 0.6899483799934387,
698
+ "eval_mean_token_accuracy": 0.810834194502134,
699
+ "eval_num_tokens": 1528119.0,
700
+ "eval_runtime": 87.8848,
701
+ "eval_samples_per_second": 16.158,
702
+ "eval_steps_per_second": 2.025,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.7141377851366997,
707
+ "epoch": 1.63971031985516,
708
+ "grad_norm": 0.8391667008399963,
709
+ "learning_rate": 0.000371667181972466,
710
+ "loss": 0.6481579780578614,
711
+ "mean_token_accuracy": 0.8159015104174614,
712
+ "num_tokens": 1573661.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.63971031985516,
717
+ "eval_entropy": 0.7598993319473909,
718
+ "eval_loss": 0.681204080581665,
719
+ "eval_mean_token_accuracy": 0.8100194586126992,
720
+ "eval_num_tokens": 1573661.0,
721
+ "eval_runtime": 88.2415,
722
+ "eval_samples_per_second": 16.092,
723
+ "eval_steps_per_second": 2.017,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.7235980287194252,
728
+ "epoch": 1.687990343995172,
729
+ "grad_norm": 0.8197196125984192,
730
+ "learning_rate": 0.00037094419338719537,
731
+ "loss": 0.6582849025726318,
732
+ "mean_token_accuracy": 0.8114834539592266,
733
+ "num_tokens": 1621714.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.687990343995172,
738
+ "eval_entropy": 0.7166854568411795,
739
+ "eval_loss": 0.679177463054657,
740
+ "eval_mean_token_accuracy": 0.8119039877077167,
741
+ "eval_num_tokens": 1621714.0,
742
+ "eval_runtime": 88.0957,
743
+ "eval_samples_per_second": 16.119,
744
+ "eval_steps_per_second": 2.021,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.7278190158307553,
749
+ "epoch": 1.736270368135184,
750
+ "grad_norm": 0.7820602655410767,
751
+ "learning_rate": 0.0003701694743306164,
752
+ "loss": 0.6633153438568116,
753
+ "mean_token_accuracy": 0.8123249113559723,
754
+ "num_tokens": 1672118.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.736270368135184,
759
+ "eval_entropy": 0.6980850314826108,
760
+ "eval_loss": 0.6775653958320618,
761
+ "eval_mean_token_accuracy": 0.8116872906684875,
762
+ "eval_num_tokens": 1672118.0,
763
+ "eval_runtime": 87.9795,
764
+ "eval_samples_per_second": 16.14,
765
+ "eval_steps_per_second": 2.023,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.7149631556123495,
770
+ "epoch": 1.7845503922751962,
771
+ "grad_norm": 1.0139760971069336,
772
+ "learning_rate": 0.00036934324403905795,
773
+ "loss": 0.6468976020812989,
774
+ "mean_token_accuracy": 0.8117561548948288,
775
+ "num_tokens": 1720465.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.7845503922751962,
780
+ "eval_entropy": 0.69009849094273,
781
+ "eval_loss": 0.6731093525886536,
782
+ "eval_mean_token_accuracy": 0.81281964196248,
783
+ "eval_num_tokens": 1720465.0,
784
+ "eval_runtime": 87.7412,
785
+ "eval_samples_per_second": 16.184,
786
+ "eval_steps_per_second": 2.029,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.7226049453020096,
791
+ "epoch": 1.832830416415208,
792
+ "grad_norm": 1.5040546655654907,
793
+ "learning_rate": 0.0003684657363259193,
794
+ "loss": 0.6385328769683838,
795
+ "mean_token_accuracy": 0.8132639616727829,
796
+ "num_tokens": 1764440.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.832830416415208,
801
+ "eval_entropy": 0.7125169618076153,
802
+ "eval_loss": 0.6864680647850037,
803
+ "eval_mean_token_accuracy": 0.8049418089095126,
804
+ "eval_num_tokens": 1764440.0,
805
+ "eval_runtime": 87.9429,
806
+ "eval_samples_per_second": 16.147,
807
+ "eval_steps_per_second": 2.024,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.707821810990572,
812
+ "epoch": 1.8811104405552204,
813
+ "grad_norm": 0.8604223132133484,
814
+ "learning_rate": 0.00036753719951550327,
815
+ "loss": 0.6397049427032471,
816
+ "mean_token_accuracy": 0.8137877315282822,
817
+ "num_tokens": 1810828.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.8811104405552204,
822
+ "eval_entropy": 0.6492102551326323,
823
+ "eval_loss": 0.6743726134300232,
824
+ "eval_mean_token_accuracy": 0.8130356572317273,
825
+ "eval_num_tokens": 1810828.0,
826
+ "eval_runtime": 87.8684,
827
+ "eval_samples_per_second": 16.161,
828
+ "eval_steps_per_second": 2.026,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.6800355084240437,
833
+ "epoch": 1.9293904646952322,
834
+ "grad_norm": 0.7270982265472412,
835
+ "learning_rate": 0.00036655789637274377,
836
+ "loss": 0.6267755031585693,
837
+ "mean_token_accuracy": 0.8180945307016373,
838
+ "num_tokens": 1857600.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9293904646952322,
843
+ "eval_entropy": 0.6727538651294922,
844
+ "eval_loss": 0.6696028709411621,
845
+ "eval_mean_token_accuracy": 0.8116443090224534,
846
+ "eval_num_tokens": 1857600.0,
847
+ "eval_runtime": 88.2449,
848
+ "eval_samples_per_second": 16.092,
849
+ "eval_steps_per_second": 2.017,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.6746691606938839,
854
+ "epoch": 1.9776704888352445,
855
+ "grad_norm": 0.9771540760993958,
856
+ "learning_rate": 0.0003655281040288461,
857
+ "loss": 0.6326992988586426,
858
+ "mean_token_accuracy": 0.8179976396262646,
859
+ "num_tokens": 1901998.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 1.9776704888352445,
864
+ "eval_entropy": 0.6636319716324967,
865
+ "eval_loss": 0.6714363098144531,
866
+ "eval_mean_token_accuracy": 0.8126635943235976,
867
+ "eval_num_tokens": 1901998.0,
868
+ "eval_runtime": 87.7666,
869
+ "eval_samples_per_second": 16.179,
870
+ "eval_steps_per_second": 2.028,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.6319125778489298,
875
+ "epoch": 2.024140012070006,
876
+ "grad_norm": 0.8380444645881653,
877
+ "learning_rate": 0.0003644481139028622,
878
+ "loss": 0.5780903816223144,
879
+ "mean_token_accuracy": 0.8250188401767186,
880
+ "num_tokens": 1947553.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.024140012070006,
885
+ "eval_entropy": 0.6627151989535,
886
+ "eval_loss": 0.6774935722351074,
887
+ "eval_mean_token_accuracy": 0.8140510819601209,
888
+ "eval_num_tokens": 1947553.0,
889
+ "eval_runtime": 87.8287,
890
+ "eval_samples_per_second": 16.168,
891
+ "eval_steps_per_second": 2.027,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.6059147048741579,
896
+ "epoch": 2.0724200362100182,
897
+ "grad_norm": 0.9745152592658997,
898
+ "learning_rate": 0.0003633182316192229,
899
+ "loss": 0.546702527999878,
900
+ "mean_token_accuracy": 0.8358186542987823,
901
+ "num_tokens": 1995234.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.0724200362100182,
906
+ "eval_entropy": 0.6620710384979677,
907
+ "eval_loss": 0.6769291162490845,
908
+ "eval_mean_token_accuracy": 0.8135082835561773,
909
+ "eval_num_tokens": 1995234.0,
910
+ "eval_runtime": 88.0008,
911
+ "eval_samples_per_second": 16.136,
912
+ "eval_steps_per_second": 2.023,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.6076117843389511,
917
+ "epoch": 2.12070006035003,
918
+ "grad_norm": 0.6914380192756653,
919
+ "learning_rate": 0.0003621387769212491,
920
+ "loss": 0.5324535846710206,
921
+ "mean_token_accuracy": 0.8366568259894848,
922
+ "num_tokens": 2040639.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.12070006035003,
927
+ "eval_entropy": 0.6592224213514435,
928
+ "eval_loss": 0.670383632183075,
929
+ "eval_mean_token_accuracy": 0.8162150088320957,
930
+ "eval_num_tokens": 2040639.0,
931
+ "eval_runtime": 88.324,
932
+ "eval_samples_per_second": 16.077,
933
+ "eval_steps_per_second": 2.015,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.6158072650432587,
938
+ "epoch": 2.1689800844900424,
939
+ "grad_norm": 0.9649198055267334,
940
+ "learning_rate": 0.0003609100835806688,
941
+ "loss": 0.5564568996429443,
942
+ "mean_token_accuracy": 0.8329654954373836,
943
+ "num_tokens": 2090604.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.1689800844900424,
948
+ "eval_entropy": 0.608588819758276,
949
+ "eval_loss": 0.6745102405548096,
950
+ "eval_mean_token_accuracy": 0.8155080542135774,
951
+ "eval_num_tokens": 2090604.0,
952
+ "eval_runtime": 87.52,
953
+ "eval_samples_per_second": 16.225,
954
+ "eval_steps_per_second": 2.034,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.5938734326511621,
959
+ "epoch": 2.2172601086300543,
960
+ "grad_norm": 0.6915557384490967,
961
+ "learning_rate": 0.0003596324993031632,
962
+ "loss": 0.5351875782012939,
963
+ "mean_token_accuracy": 0.8385631121695042,
964
+ "num_tokens": 2142246.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2172601086300543,
969
+ "eval_entropy": 0.6251631205001574,
970
+ "eval_loss": 0.6664518117904663,
971
+ "eval_mean_token_accuracy": 0.81711940417129,
972
+ "eval_num_tokens": 2142246.0,
973
+ "eval_runtime": 87.1792,
974
+ "eval_samples_per_second": 16.288,
975
+ "eval_steps_per_second": 2.042,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.623173613846302,
980
+ "epoch": 2.2655401327700666,
981
+ "grad_norm": 0.8366610407829285,
982
+ "learning_rate": 0.00035830638562997063,
983
+ "loss": 0.5488213539123535,
984
+ "mean_token_accuracy": 0.8321483485400677,
985
+ "num_tokens": 2183144.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.2655401327700666,
990
+ "eval_entropy": 0.6237085649136747,
991
+ "eval_loss": 0.68075031042099,
992
+ "eval_mean_token_accuracy": 0.814671738429016,
993
+ "eval_num_tokens": 2183144.0,
994
+ "eval_runtime": 88.0652,
995
+ "eval_samples_per_second": 16.124,
996
+ "eval_steps_per_second": 2.021,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.6236911740154027,
1001
+ "epoch": 2.3138201569100785,
1002
+ "grad_norm": 0.7812452912330627,
1003
+ "learning_rate": 0.00035693211783557416,
1004
+ "loss": 0.5696200847625732,
1005
+ "mean_token_accuracy": 0.82991396561265,
1006
+ "num_tokens": 2228098.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3138201569100785,
1011
+ "eval_entropy": 0.6327033426319615,
1012
+ "eval_loss": 0.6742984652519226,
1013
+ "eval_mean_token_accuracy": 0.8148086091105857,
1014
+ "eval_num_tokens": 2228098.0,
1015
+ "eval_runtime": 88.1711,
1016
+ "eval_samples_per_second": 16.105,
1017
+ "eval_steps_per_second": 2.019,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.6218426413834095,
1022
+ "epoch": 2.3621001810500903,
1023
+ "grad_norm": 0.8719790577888489,
1024
+ "learning_rate": 0.0003555100848215035,
1025
+ "loss": 0.5582651615142822,
1026
+ "mean_token_accuracy": 0.8323961742222309,
1027
+ "num_tokens": 2273062.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.3621001810500903,
1032
+ "eval_entropy": 0.6338012375858393,
1033
+ "eval_loss": 0.6685822606086731,
1034
+ "eval_mean_token_accuracy": 0.8152180300669724,
1035
+ "eval_num_tokens": 2273062.0,
1036
+ "eval_runtime": 88.101,
1037
+ "eval_samples_per_second": 16.118,
1038
+ "eval_steps_per_second": 2.02,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.6050681818276644,
1043
+ "epoch": 2.4103802051901027,
1044
+ "grad_norm": 0.9151566028594971,
1045
+ "learning_rate": 0.00035404068900628076,
1046
+ "loss": 0.559452486038208,
1047
+ "mean_token_accuracy": 0.8310242861509323,
1048
+ "num_tokens": 2319305.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.4103802051901027,
1053
+ "eval_entropy": 0.6634890969549672,
1054
+ "eval_loss": 0.6692465543746948,
1055
+ "eval_mean_token_accuracy": 0.8134353673190213,
1056
+ "eval_num_tokens": 2319305.0,
1057
+ "eval_runtime": 87.7711,
1058
+ "eval_samples_per_second": 16.178,
1059
+ "eval_steps_per_second": 2.028,
1060
+ "step": 1000
1061
+ }
1062
+ ],
1063
+ "logging_steps": 20,
1064
+ "max_steps": 4150,
1065
+ "num_input_tokens_seen": 0,
1066
+ "num_train_epochs": 10,
1067
+ "save_steps": 20,
1068
+ "stateful_callbacks": {
1069
+ "TrainerControl": {
1070
+ "args": {
1071
+ "should_epoch_stop": false,
1072
+ "should_evaluate": false,
1073
+ "should_log": false,
1074
+ "should_save": true,
1075
+ "should_training_stop": false
1076
+ },
1077
+ "attributes": {}
1078
+ }
1079
+ },
1080
+ "total_flos": 9.8038608775637e+16,
1081
+ "train_batch_size": 4,
1082
+ "trial_name": null,
1083
+ "trial_params": null
1084
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.03828026524568501,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "v_proj",
34
+ "up_proj",
35
+ "down_proj",
36
+ "o_proj",
37
+ "q_proj",
38
+ "k_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/trainer_state.json ADDED
@@ -0,0 +1,1105 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.4586602293301145,
6
+ "eval_steps": 20,
7
+ "global_step": 1020,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 2.172999547421932,
14
+ "epoch": 0.04828002414001207,
15
+ "grad_norm": 2.5071208477020264,
16
+ "learning_rate": 1.7227585487708594e-05,
17
+ "loss": 1.9347333908081055,
18
+ "mean_token_accuracy": 0.6086816020309925,
19
+ "num_tokens": 47778.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.04828002414001207,
24
+ "eval_entropy": 1.6909754791956269,
25
+ "eval_loss": 1.4728385210037231,
26
+ "eval_mean_token_accuracy": 0.6773593797442619,
27
+ "eval_num_tokens": 47778.0,
28
+ "eval_runtime": 90.1265,
29
+ "eval_samples_per_second": 15.756,
30
+ "eval_steps_per_second": 1.975,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.2014626950025558,
35
+ "epoch": 0.09656004828002414,
36
+ "grad_norm": 1.5074543952941895,
37
+ "learning_rate": 3.5361886001086065e-05,
38
+ "loss": 1.0867146492004394,
39
+ "mean_token_accuracy": 0.7285927519202232,
40
+ "num_tokens": 95981.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.09656004828002414,
45
+ "eval_entropy": 1.0304168352250302,
46
+ "eval_loss": 0.9211422204971313,
47
+ "eval_mean_token_accuracy": 0.7613122848312507,
48
+ "eval_num_tokens": 95981.0,
49
+ "eval_runtime": 88.6338,
50
+ "eval_samples_per_second": 16.021,
51
+ "eval_steps_per_second": 2.008,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.9454198732972146,
56
+ "epoch": 0.14484007242003621,
57
+ "grad_norm": 1.7578108310699463,
58
+ "learning_rate": 5.349618651446354e-05,
59
+ "loss": 0.8447297096252442,
60
+ "mean_token_accuracy": 0.7723424732685089,
61
+ "num_tokens": 142784.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.14484007242003621,
66
+ "eval_entropy": 0.8647834474451086,
67
+ "eval_loss": 0.8240737318992615,
68
+ "eval_mean_token_accuracy": 0.7821227014064789,
69
+ "eval_num_tokens": 142784.0,
70
+ "eval_runtime": 88.0135,
71
+ "eval_samples_per_second": 16.134,
72
+ "eval_steps_per_second": 2.022,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.8963550612330436,
77
+ "epoch": 0.19312009656004828,
78
+ "grad_norm": 1.1464931964874268,
79
+ "learning_rate": 7.1630487027841e-05,
80
+ "loss": 0.7979836463928223,
81
+ "mean_token_accuracy": 0.7822489373385906,
82
+ "num_tokens": 187338.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.19312009656004828,
87
+ "eval_entropy": 0.8663296133614657,
88
+ "eval_loss": 0.7752988338470459,
89
+ "eval_mean_token_accuracy": 0.7885957012015782,
90
+ "eval_num_tokens": 187338.0,
91
+ "eval_runtime": 88.284,
92
+ "eval_samples_per_second": 16.084,
93
+ "eval_steps_per_second": 2.016,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.8265103206038475,
98
+ "epoch": 0.24140012070006034,
99
+ "grad_norm": 1.0106662511825562,
100
+ "learning_rate": 8.976478754121848e-05,
101
+ "loss": 0.7297060012817382,
102
+ "mean_token_accuracy": 0.7955138415098191,
103
+ "num_tokens": 236038.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24140012070006034,
108
+ "eval_entropy": 0.8122169536151244,
109
+ "eval_loss": 0.754906177520752,
110
+ "eval_mean_token_accuracy": 0.7917155255092664,
111
+ "eval_num_tokens": 236038.0,
112
+ "eval_runtime": 87.5754,
113
+ "eval_samples_per_second": 16.215,
114
+ "eval_steps_per_second": 2.033,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.8098392844200134,
119
+ "epoch": 0.28968014484007243,
120
+ "grad_norm": 1.2238041162490845,
121
+ "learning_rate": 0.00010789908805459595,
122
+ "loss": 0.7201489448547364,
123
+ "mean_token_accuracy": 0.7975016921758652,
124
+ "num_tokens": 282918.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.28968014484007243,
129
+ "eval_entropy": 0.8418726462326692,
130
+ "eval_loss": 0.7393178939819336,
131
+ "eval_mean_token_accuracy": 0.7936771585700217,
132
+ "eval_num_tokens": 282918.0,
133
+ "eval_runtime": 87.1597,
134
+ "eval_samples_per_second": 16.292,
135
+ "eval_steps_per_second": 2.042,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.8080591082572937,
140
+ "epoch": 0.33796016898008446,
141
+ "grad_norm": 1.011591911315918,
142
+ "learning_rate": 0.00012603338856797343,
143
+ "loss": 0.7147697925567627,
144
+ "mean_token_accuracy": 0.7947175532579422,
145
+ "num_tokens": 331305.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.33796016898008446,
150
+ "eval_entropy": 0.7894941431083037,
151
+ "eval_loss": 0.720219075679779,
152
+ "eval_mean_token_accuracy": 0.801899236239744,
153
+ "eval_num_tokens": 331305.0,
154
+ "eval_runtime": 87.1493,
155
+ "eval_samples_per_second": 16.294,
156
+ "eval_steps_per_second": 2.042,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.7957205109298229,
161
+ "epoch": 0.38624019312009655,
162
+ "grad_norm": 0.8481830954551697,
163
+ "learning_rate": 0.00014416768908135088,
164
+ "loss": 0.7050027847290039,
165
+ "mean_token_accuracy": 0.7995368830859662,
166
+ "num_tokens": 375876.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.38624019312009655,
171
+ "eval_entropy": 0.7746140591883928,
172
+ "eval_loss": 0.7175475358963013,
173
+ "eval_mean_token_accuracy": 0.7994848955213354,
174
+ "eval_num_tokens": 375876.0,
175
+ "eval_runtime": 86.655,
176
+ "eval_samples_per_second": 16.387,
177
+ "eval_steps_per_second": 2.054,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.779565304517746,
182
+ "epoch": 0.43452021726010864,
183
+ "grad_norm": 1.032624363899231,
184
+ "learning_rate": 0.00016230198959472835,
185
+ "loss": 0.6978332042694092,
186
+ "mean_token_accuracy": 0.8015618294477462,
187
+ "num_tokens": 423913.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.43452021726010864,
192
+ "eval_entropy": 0.7455583871080634,
193
+ "eval_loss": 0.7118850350379944,
194
+ "eval_mean_token_accuracy": 0.7999682195400923,
195
+ "eval_num_tokens": 423913.0,
196
+ "eval_runtime": 86.7842,
197
+ "eval_samples_per_second": 16.362,
198
+ "eval_steps_per_second": 2.051,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.7810172818601131,
203
+ "epoch": 0.4828002414001207,
204
+ "grad_norm": 0.9276663064956665,
205
+ "learning_rate": 0.00018043629010810582,
206
+ "loss": 0.6937861442565918,
207
+ "mean_token_accuracy": 0.802893053740263,
208
+ "num_tokens": 472395.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.4828002414001207,
213
+ "eval_entropy": 0.7650193290764027,
214
+ "eval_loss": 0.7020567059516907,
215
+ "eval_mean_token_accuracy": 0.8045302153973097,
216
+ "eval_num_tokens": 472395.0,
217
+ "eval_runtime": 88.326,
218
+ "eval_samples_per_second": 16.077,
219
+ "eval_steps_per_second": 2.015,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.7862060204148292,
224
+ "epoch": 0.5310802655401328,
225
+ "grad_norm": 0.8195295929908752,
226
+ "learning_rate": 0.0001985705906214833,
227
+ "loss": 0.6878782272338867,
228
+ "mean_token_accuracy": 0.8027168907225132,
229
+ "num_tokens": 517049.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.5310802655401328,
234
+ "eval_entropy": 0.7656565583154057,
235
+ "eval_loss": 0.6973932981491089,
236
+ "eval_mean_token_accuracy": 0.8047967278555538,
237
+ "eval_num_tokens": 517049.0,
238
+ "eval_runtime": 87.7698,
239
+ "eval_samples_per_second": 16.179,
240
+ "eval_steps_per_second": 2.028,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.7611534461379051,
245
+ "epoch": 0.5793602896801449,
246
+ "grad_norm": 0.7499712109565735,
247
+ "learning_rate": 0.00021670489113486077,
248
+ "loss": 0.681937837600708,
249
+ "mean_token_accuracy": 0.8041978515684605,
250
+ "num_tokens": 562836.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.5793602896801449,
255
+ "eval_entropy": 0.7599442380197933,
256
+ "eval_loss": 0.7002046704292297,
257
+ "eval_mean_token_accuracy": 0.8051861517884759,
258
+ "eval_num_tokens": 562836.0,
259
+ "eval_runtime": 87.9239,
260
+ "eval_samples_per_second": 16.15,
261
+ "eval_steps_per_second": 2.024,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.7800783462822437,
266
+ "epoch": 0.627640313820157,
267
+ "grad_norm": 0.8645269274711609,
268
+ "learning_rate": 0.00023483919164823825,
269
+ "loss": 0.6954500675201416,
270
+ "mean_token_accuracy": 0.8031484372913837,
271
+ "num_tokens": 606612.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.627640313820157,
276
+ "eval_entropy": 0.7498679090751691,
277
+ "eval_loss": 0.7046149373054504,
278
+ "eval_mean_token_accuracy": 0.802388215667746,
279
+ "eval_num_tokens": 606612.0,
280
+ "eval_runtime": 88.2468,
281
+ "eval_samples_per_second": 16.091,
282
+ "eval_steps_per_second": 2.017,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.7586533665657044,
287
+ "epoch": 0.6759203379601689,
288
+ "grad_norm": 0.741596519947052,
289
+ "learning_rate": 0.0002529734921616157,
290
+ "loss": 0.690476369857788,
291
+ "mean_token_accuracy": 0.8046923100948333,
292
+ "num_tokens": 655920.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6759203379601689,
297
+ "eval_entropy": 0.7714274066887544,
298
+ "eval_loss": 0.6992160677909851,
299
+ "eval_mean_token_accuracy": 0.8042646482419432,
300
+ "eval_num_tokens": 655920.0,
301
+ "eval_runtime": 88.1754,
302
+ "eval_samples_per_second": 16.104,
303
+ "eval_steps_per_second": 2.019,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.7585848286747933,
308
+ "epoch": 0.724200362100181,
309
+ "grad_norm": 1.0778124332427979,
310
+ "learning_rate": 0.00027110779267499317,
311
+ "loss": 0.6790409564971924,
312
+ "mean_token_accuracy": 0.8060351841151714,
313
+ "num_tokens": 702638.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.724200362100181,
318
+ "eval_entropy": 0.7380402811457601,
319
+ "eval_loss": 0.7054820656776428,
320
+ "eval_mean_token_accuracy": 0.8003534738267406,
321
+ "eval_num_tokens": 702638.0,
322
+ "eval_runtime": 88.238,
323
+ "eval_samples_per_second": 16.093,
324
+ "eval_steps_per_second": 2.017,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.7696981988847256,
329
+ "epoch": 0.7724803862401931,
330
+ "grad_norm": 0.8408658504486084,
331
+ "learning_rate": 0.00028924209318837064,
332
+ "loss": 0.7070387363433838,
333
+ "mean_token_accuracy": 0.8035947173833847,
334
+ "num_tokens": 749377.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.7724803862401931,
339
+ "eval_entropy": 0.7897409023193831,
340
+ "eval_loss": 0.7015668749809265,
341
+ "eval_mean_token_accuracy": 0.805306687113944,
342
+ "eval_num_tokens": 749377.0,
343
+ "eval_runtime": 87.9395,
344
+ "eval_samples_per_second": 16.147,
345
+ "eval_steps_per_second": 2.024,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.762337189912796,
350
+ "epoch": 0.8207604103802052,
351
+ "grad_norm": 0.9843310117721558,
352
+ "learning_rate": 0.0003073763937017481,
353
+ "loss": 0.7021088123321533,
354
+ "mean_token_accuracy": 0.8029616877436638,
355
+ "num_tokens": 798874.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8207604103802052,
360
+ "eval_entropy": 0.7210077121016685,
361
+ "eval_loss": 0.7082746624946594,
362
+ "eval_mean_token_accuracy": 0.8049283824609906,
363
+ "eval_num_tokens": 798874.0,
364
+ "eval_runtime": 88.1706,
365
+ "eval_samples_per_second": 16.105,
366
+ "eval_steps_per_second": 2.019,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.7879025422036647,
371
+ "epoch": 0.8690404345202173,
372
+ "grad_norm": 0.9713814854621887,
373
+ "learning_rate": 0.0003255106942151256,
374
+ "loss": 0.714624547958374,
375
+ "mean_token_accuracy": 0.79987031519413,
376
+ "num_tokens": 840825.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8690404345202173,
381
+ "eval_entropy": 0.7866930489459735,
382
+ "eval_loss": 0.7145519852638245,
383
+ "eval_mean_token_accuracy": 0.8015657988157165,
384
+ "eval_num_tokens": 840825.0,
385
+ "eval_runtime": 88.3746,
386
+ "eval_samples_per_second": 16.068,
387
+ "eval_steps_per_second": 2.014,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.7844726584851742,
392
+ "epoch": 0.9173204586602294,
393
+ "grad_norm": 1.2028311491012573,
394
+ "learning_rate": 0.00034364499472850306,
395
+ "loss": 0.7040786266326904,
396
+ "mean_token_accuracy": 0.7981885217130185,
397
+ "num_tokens": 882885.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9173204586602294,
402
+ "eval_entropy": 0.7767668339643585,
403
+ "eval_loss": 0.7073574066162109,
404
+ "eval_mean_token_accuracy": 0.8040851833445303,
405
+ "eval_num_tokens": 882885.0,
406
+ "eval_runtime": 88.3828,
407
+ "eval_samples_per_second": 16.066,
408
+ "eval_steps_per_second": 2.014,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.7699790760874748,
413
+ "epoch": 0.9656004828002414,
414
+ "grad_norm": 0.9397398233413696,
415
+ "learning_rate": 0.00036177929524188054,
416
+ "loss": 0.7040313720703125,
417
+ "mean_token_accuracy": 0.8034727744758129,
418
+ "num_tokens": 927456.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9656004828002414,
423
+ "eval_entropy": 0.7749103545472863,
424
+ "eval_loss": 0.7093513011932373,
425
+ "eval_mean_token_accuracy": 0.8019482901926791,
426
+ "eval_num_tokens": 927456.0,
427
+ "eval_runtime": 86.2112,
428
+ "eval_samples_per_second": 16.471,
429
+ "eval_steps_per_second": 2.065,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.7672389172888422,
434
+ "epoch": 1.012070006035003,
435
+ "grad_norm": 0.99210524559021,
436
+ "learning_rate": 0.00037628567078152296,
437
+ "loss": 0.6913118362426758,
438
+ "mean_token_accuracy": 0.8042210944287189,
439
+ "num_tokens": 972662.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.012070006035003,
444
+ "eval_entropy": 0.7294771229283193,
445
+ "eval_loss": 0.7171520590782166,
446
+ "eval_mean_token_accuracy": 0.8015128585059991,
447
+ "eval_num_tokens": 972662.0,
448
+ "eval_runtime": 88.037,
449
+ "eval_samples_per_second": 16.13,
450
+ "eval_steps_per_second": 2.022,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.7143943183124065,
455
+ "epoch": 1.060350030175015,
456
+ "grad_norm": 1.087638258934021,
457
+ "learning_rate": 0.0003762484015601069,
458
+ "loss": 0.642704439163208,
459
+ "mean_token_accuracy": 0.8124966286122799,
460
+ "num_tokens": 1022525.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.060350030175015,
465
+ "eval_entropy": 0.743511886744017,
466
+ "eval_loss": 0.7116662859916687,
467
+ "eval_mean_token_accuracy": 0.8037210672758939,
468
+ "eval_num_tokens": 1022525.0,
469
+ "eval_runtime": 87.9931,
470
+ "eval_samples_per_second": 16.138,
471
+ "eval_steps_per_second": 2.023,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.7237346574664116,
476
+ "epoch": 1.1086300543150271,
477
+ "grad_norm": 0.9365553259849548,
478
+ "learning_rate": 0.0003761579008392123,
479
+ "loss": 0.652644681930542,
480
+ "mean_token_accuracy": 0.8107496216893196,
481
+ "num_tokens": 1069194.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1086300543150271,
486
+ "eval_entropy": 0.7372823839776972,
487
+ "eval_loss": 0.7238303422927856,
488
+ "eval_mean_token_accuracy": 0.8026230124275336,
489
+ "eval_num_tokens": 1069194.0,
490
+ "eval_runtime": 87.6778,
491
+ "eval_samples_per_second": 16.196,
492
+ "eval_steps_per_second": 2.03,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.7202137842774391,
497
+ "epoch": 1.1569100784550392,
498
+ "grad_norm": 0.7998443841934204,
499
+ "learning_rate": 0.0003760141942294726,
500
+ "loss": 0.6485219478607178,
501
+ "mean_token_accuracy": 0.8092714451253414,
502
+ "num_tokens": 1117570.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1569100784550392,
507
+ "eval_entropy": 0.71547647942318,
508
+ "eval_loss": 0.7124277949333191,
509
+ "eval_mean_token_accuracy": 0.8013174219077892,
510
+ "eval_num_tokens": 1117570.0,
511
+ "eval_runtime": 87.4387,
512
+ "eval_samples_per_second": 16.24,
513
+ "eval_steps_per_second": 2.036,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.7270819500088692,
518
+ "epoch": 1.2051901025950513,
519
+ "grad_norm": 0.9470787644386292,
520
+ "learning_rate": 0.00037581732239815854,
521
+ "loss": 0.66352219581604,
522
+ "mean_token_accuracy": 0.8102593503892421,
523
+ "num_tokens": 1161717.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.2051901025950513,
528
+ "eval_entropy": 0.7791565897759427,
529
+ "eval_loss": 0.7179726958274841,
530
+ "eval_mean_token_accuracy": 0.7948898328154275,
531
+ "eval_num_tokens": 1161717.0,
532
+ "eval_runtime": 87.9927,
533
+ "eval_samples_per_second": 16.138,
534
+ "eval_steps_per_second": 2.023,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.7320936664938926,
539
+ "epoch": 1.2534701267350634,
540
+ "grad_norm": 1.0220379829406738,
541
+ "learning_rate": 0.0003755673410576695,
542
+ "loss": 0.6581549644470215,
543
+ "mean_token_accuracy": 0.8098709337413311,
544
+ "num_tokens": 1209390.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.2534701267350634,
549
+ "eval_entropy": 0.7710678225822662,
550
+ "eval_loss": 0.708797812461853,
551
+ "eval_mean_token_accuracy": 0.8039572778042783,
552
+ "eval_num_tokens": 1209390.0,
553
+ "eval_runtime": 88.1193,
554
+ "eval_samples_per_second": 16.115,
555
+ "eval_steps_per_second": 2.02,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.7281674664467573,
560
+ "epoch": 1.3017501508750755,
561
+ "grad_norm": 0.8775188326835632,
562
+ "learning_rate": 0.000375264320949768,
563
+ "loss": 0.6631278991699219,
564
+ "mean_token_accuracy": 0.8101261422038079,
565
+ "num_tokens": 1256611.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3017501508750755,
570
+ "eval_entropy": 0.7199799800187014,
571
+ "eval_loss": 0.6981531381607056,
572
+ "eval_mean_token_accuracy": 0.8071265254127845,
573
+ "eval_num_tokens": 1256611.0,
574
+ "eval_runtime": 88.165,
575
+ "eval_samples_per_second": 16.106,
576
+ "eval_steps_per_second": 2.019,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.722860773652792,
581
+ "epoch": 1.3500301750150876,
582
+ "grad_norm": 0.8949288129806519,
583
+ "learning_rate": 0.00037490834782556,
584
+ "loss": 0.6639598846435547,
585
+ "mean_token_accuracy": 0.8105649061501026,
586
+ "num_tokens": 1302325.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3500301750150876,
591
+ "eval_entropy": 0.7134760032209118,
592
+ "eval_loss": 0.7029948234558105,
593
+ "eval_mean_token_accuracy": 0.8062985880991046,
594
+ "eval_num_tokens": 1302325.0,
595
+ "eval_runtime": 88.1286,
596
+ "eval_samples_per_second": 16.113,
597
+ "eval_steps_per_second": 2.02,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.7299126699566841,
602
+ "epoch": 1.3983101991550995,
603
+ "grad_norm": 1.0323039293289185,
604
+ "learning_rate": 0.0003744995224212291,
605
+ "loss": 0.6535763263702392,
606
+ "mean_token_accuracy": 0.8120224848389626,
607
+ "num_tokens": 1343844.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.3983101991550995,
612
+ "eval_entropy": 0.7169701182440426,
613
+ "eval_loss": 0.7067857980728149,
614
+ "eval_mean_token_accuracy": 0.8040432936689826,
615
+ "eval_num_tokens": 1343844.0,
616
+ "eval_runtime": 88.2252,
617
+ "eval_samples_per_second": 16.095,
618
+ "eval_steps_per_second": 2.018,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.7281742256134749,
623
+ "epoch": 1.4465902232951118,
624
+ "grad_norm": 1.3370424509048462,
625
+ "learning_rate": 0.00037403796042952863,
626
+ "loss": 0.6637272357940673,
627
+ "mean_token_accuracy": 0.8118261776864528,
628
+ "num_tokens": 1389471.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4465902232951118,
633
+ "eval_entropy": 0.7378843212395572,
634
+ "eval_loss": 0.7038390040397644,
635
+ "eval_mean_token_accuracy": 0.8059229371922739,
636
+ "eval_num_tokens": 1389471.0,
637
+ "eval_runtime": 87.9301,
638
+ "eval_samples_per_second": 16.149,
639
+ "eval_steps_per_second": 2.024,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.7388614989817143,
644
+ "epoch": 1.4948702474351236,
645
+ "grad_norm": 0.8464171290397644,
646
+ "learning_rate": 0.00037352379246704257,
647
+ "loss": 0.6604740142822265,
648
+ "mean_token_accuracy": 0.8102974124252796,
649
+ "num_tokens": 1435975.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.4948702474351236,
654
+ "eval_entropy": 0.7126847046814607,
655
+ "eval_loss": 0.6950703263282776,
656
+ "eval_mean_token_accuracy": 0.8085885392815879,
657
+ "eval_num_tokens": 1435975.0,
658
+ "eval_runtime": 88.2732,
659
+ "eval_samples_per_second": 16.086,
660
+ "eval_steps_per_second": 2.016,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.7258916139602661,
665
+ "epoch": 1.5431502715751357,
666
+ "grad_norm": 1.1780085563659668,
667
+ "learning_rate": 0.0003729571640372223,
668
+ "loss": 0.6529891014099121,
669
+ "mean_token_accuracy": 0.8085981778800487,
670
+ "num_tokens": 1482035.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.5431502715751357,
675
+ "eval_entropy": 0.7096509638797032,
676
+ "eval_loss": 0.690051257610321,
677
+ "eval_mean_token_accuracy": 0.8100520241796301,
678
+ "eval_num_tokens": 1482035.0,
679
+ "eval_runtime": 88.3153,
680
+ "eval_samples_per_second": 16.079,
681
+ "eval_steps_per_second": 2.016,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.7151173129677773,
686
+ "epoch": 1.5914302957151478,
687
+ "grad_norm": 1.0357667207717896,
688
+ "learning_rate": 0.0003723382354892108,
689
+ "loss": 0.6451474189758301,
690
+ "mean_token_accuracy": 0.8136902332305909,
691
+ "num_tokens": 1528119.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.5914302957151478,
696
+ "eval_entropy": 0.7060252652409371,
697
+ "eval_loss": 0.6899483799934387,
698
+ "eval_mean_token_accuracy": 0.810834194502134,
699
+ "eval_num_tokens": 1528119.0,
700
+ "eval_runtime": 87.8848,
701
+ "eval_samples_per_second": 16.158,
702
+ "eval_steps_per_second": 2.025,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.7141377851366997,
707
+ "epoch": 1.63971031985516,
708
+ "grad_norm": 0.8391667008399963,
709
+ "learning_rate": 0.000371667181972466,
710
+ "loss": 0.6481579780578614,
711
+ "mean_token_accuracy": 0.8159015104174614,
712
+ "num_tokens": 1573661.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.63971031985516,
717
+ "eval_entropy": 0.7598993319473909,
718
+ "eval_loss": 0.681204080581665,
719
+ "eval_mean_token_accuracy": 0.8100194586126992,
720
+ "eval_num_tokens": 1573661.0,
721
+ "eval_runtime": 88.2415,
722
+ "eval_samples_per_second": 16.092,
723
+ "eval_steps_per_second": 2.017,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.7235980287194252,
728
+ "epoch": 1.687990343995172,
729
+ "grad_norm": 0.8197196125984192,
730
+ "learning_rate": 0.00037094419338719537,
731
+ "loss": 0.6582849025726318,
732
+ "mean_token_accuracy": 0.8114834539592266,
733
+ "num_tokens": 1621714.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.687990343995172,
738
+ "eval_entropy": 0.7166854568411795,
739
+ "eval_loss": 0.679177463054657,
740
+ "eval_mean_token_accuracy": 0.8119039877077167,
741
+ "eval_num_tokens": 1621714.0,
742
+ "eval_runtime": 88.0957,
743
+ "eval_samples_per_second": 16.119,
744
+ "eval_steps_per_second": 2.021,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.7278190158307553,
749
+ "epoch": 1.736270368135184,
750
+ "grad_norm": 0.7820602655410767,
751
+ "learning_rate": 0.0003701694743306164,
752
+ "loss": 0.6633153438568116,
753
+ "mean_token_accuracy": 0.8123249113559723,
754
+ "num_tokens": 1672118.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.736270368135184,
759
+ "eval_entropy": 0.6980850314826108,
760
+ "eval_loss": 0.6775653958320618,
761
+ "eval_mean_token_accuracy": 0.8116872906684875,
762
+ "eval_num_tokens": 1672118.0,
763
+ "eval_runtime": 87.9795,
764
+ "eval_samples_per_second": 16.14,
765
+ "eval_steps_per_second": 2.023,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.7149631556123495,
770
+ "epoch": 1.7845503922751962,
771
+ "grad_norm": 1.0139760971069336,
772
+ "learning_rate": 0.00036934324403905795,
773
+ "loss": 0.6468976020812989,
774
+ "mean_token_accuracy": 0.8117561548948288,
775
+ "num_tokens": 1720465.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.7845503922751962,
780
+ "eval_entropy": 0.69009849094273,
781
+ "eval_loss": 0.6731093525886536,
782
+ "eval_mean_token_accuracy": 0.81281964196248,
783
+ "eval_num_tokens": 1720465.0,
784
+ "eval_runtime": 87.7412,
785
+ "eval_samples_per_second": 16.184,
786
+ "eval_steps_per_second": 2.029,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.7226049453020096,
791
+ "epoch": 1.832830416415208,
792
+ "grad_norm": 1.5040546655654907,
793
+ "learning_rate": 0.0003684657363259193,
794
+ "loss": 0.6385328769683838,
795
+ "mean_token_accuracy": 0.8132639616727829,
796
+ "num_tokens": 1764440.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.832830416415208,
801
+ "eval_entropy": 0.7125169618076153,
802
+ "eval_loss": 0.6864680647850037,
803
+ "eval_mean_token_accuracy": 0.8049418089095126,
804
+ "eval_num_tokens": 1764440.0,
805
+ "eval_runtime": 87.9429,
806
+ "eval_samples_per_second": 16.147,
807
+ "eval_steps_per_second": 2.024,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.707821810990572,
812
+ "epoch": 1.8811104405552204,
813
+ "grad_norm": 0.8604223132133484,
814
+ "learning_rate": 0.00036753719951550327,
815
+ "loss": 0.6397049427032471,
816
+ "mean_token_accuracy": 0.8137877315282822,
817
+ "num_tokens": 1810828.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.8811104405552204,
822
+ "eval_entropy": 0.6492102551326323,
823
+ "eval_loss": 0.6743726134300232,
824
+ "eval_mean_token_accuracy": 0.8130356572317273,
825
+ "eval_num_tokens": 1810828.0,
826
+ "eval_runtime": 87.8684,
827
+ "eval_samples_per_second": 16.161,
828
+ "eval_steps_per_second": 2.026,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.6800355084240437,
833
+ "epoch": 1.9293904646952322,
834
+ "grad_norm": 0.7270982265472412,
835
+ "learning_rate": 0.00036655789637274377,
836
+ "loss": 0.6267755031585693,
837
+ "mean_token_accuracy": 0.8180945307016373,
838
+ "num_tokens": 1857600.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9293904646952322,
843
+ "eval_entropy": 0.6727538651294922,
844
+ "eval_loss": 0.6696028709411621,
845
+ "eval_mean_token_accuracy": 0.8116443090224534,
846
+ "eval_num_tokens": 1857600.0,
847
+ "eval_runtime": 88.2449,
848
+ "eval_samples_per_second": 16.092,
849
+ "eval_steps_per_second": 2.017,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.6746691606938839,
854
+ "epoch": 1.9776704888352445,
855
+ "grad_norm": 0.9771540760993958,
856
+ "learning_rate": 0.0003655281040288461,
857
+ "loss": 0.6326992988586426,
858
+ "mean_token_accuracy": 0.8179976396262646,
859
+ "num_tokens": 1901998.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 1.9776704888352445,
864
+ "eval_entropy": 0.6636319716324967,
865
+ "eval_loss": 0.6714363098144531,
866
+ "eval_mean_token_accuracy": 0.8126635943235976,
867
+ "eval_num_tokens": 1901998.0,
868
+ "eval_runtime": 87.7666,
869
+ "eval_samples_per_second": 16.179,
870
+ "eval_steps_per_second": 2.028,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.6319125778489298,
875
+ "epoch": 2.024140012070006,
876
+ "grad_norm": 0.8380444645881653,
877
+ "learning_rate": 0.0003644481139028622,
878
+ "loss": 0.5780903816223144,
879
+ "mean_token_accuracy": 0.8250188401767186,
880
+ "num_tokens": 1947553.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.024140012070006,
885
+ "eval_entropy": 0.6627151989535,
886
+ "eval_loss": 0.6774935722351074,
887
+ "eval_mean_token_accuracy": 0.8140510819601209,
888
+ "eval_num_tokens": 1947553.0,
889
+ "eval_runtime": 87.8287,
890
+ "eval_samples_per_second": 16.168,
891
+ "eval_steps_per_second": 2.027,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.6059147048741579,
896
+ "epoch": 2.0724200362100182,
897
+ "grad_norm": 0.9745152592658997,
898
+ "learning_rate": 0.0003633182316192229,
899
+ "loss": 0.546702527999878,
900
+ "mean_token_accuracy": 0.8358186542987823,
901
+ "num_tokens": 1995234.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.0724200362100182,
906
+ "eval_entropy": 0.6620710384979677,
907
+ "eval_loss": 0.6769291162490845,
908
+ "eval_mean_token_accuracy": 0.8135082835561773,
909
+ "eval_num_tokens": 1995234.0,
910
+ "eval_runtime": 88.0008,
911
+ "eval_samples_per_second": 16.136,
912
+ "eval_steps_per_second": 2.023,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.6076117843389511,
917
+ "epoch": 2.12070006035003,
918
+ "grad_norm": 0.6914380192756653,
919
+ "learning_rate": 0.0003621387769212491,
920
+ "loss": 0.5324535846710206,
921
+ "mean_token_accuracy": 0.8366568259894848,
922
+ "num_tokens": 2040639.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.12070006035003,
927
+ "eval_entropy": 0.6592224213514435,
928
+ "eval_loss": 0.670383632183075,
929
+ "eval_mean_token_accuracy": 0.8162150088320957,
930
+ "eval_num_tokens": 2040639.0,
931
+ "eval_runtime": 88.324,
932
+ "eval_samples_per_second": 16.077,
933
+ "eval_steps_per_second": 2.015,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.6158072650432587,
938
+ "epoch": 2.1689800844900424,
939
+ "grad_norm": 0.9649198055267334,
940
+ "learning_rate": 0.0003609100835806688,
941
+ "loss": 0.5564568996429443,
942
+ "mean_token_accuracy": 0.8329654954373836,
943
+ "num_tokens": 2090604.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.1689800844900424,
948
+ "eval_entropy": 0.608588819758276,
949
+ "eval_loss": 0.6745102405548096,
950
+ "eval_mean_token_accuracy": 0.8155080542135774,
951
+ "eval_num_tokens": 2090604.0,
952
+ "eval_runtime": 87.52,
953
+ "eval_samples_per_second": 16.225,
954
+ "eval_steps_per_second": 2.034,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.5938734326511621,
959
+ "epoch": 2.2172601086300543,
960
+ "grad_norm": 0.6915557384490967,
961
+ "learning_rate": 0.0003596324993031632,
962
+ "loss": 0.5351875782012939,
963
+ "mean_token_accuracy": 0.8385631121695042,
964
+ "num_tokens": 2142246.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2172601086300543,
969
+ "eval_entropy": 0.6251631205001574,
970
+ "eval_loss": 0.6664518117904663,
971
+ "eval_mean_token_accuracy": 0.81711940417129,
972
+ "eval_num_tokens": 2142246.0,
973
+ "eval_runtime": 87.1792,
974
+ "eval_samples_per_second": 16.288,
975
+ "eval_steps_per_second": 2.042,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.623173613846302,
980
+ "epoch": 2.2655401327700666,
981
+ "grad_norm": 0.8366610407829285,
982
+ "learning_rate": 0.00035830638562997063,
983
+ "loss": 0.5488213539123535,
984
+ "mean_token_accuracy": 0.8321483485400677,
985
+ "num_tokens": 2183144.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.2655401327700666,
990
+ "eval_entropy": 0.6237085649136747,
991
+ "eval_loss": 0.68075031042099,
992
+ "eval_mean_token_accuracy": 0.814671738429016,
993
+ "eval_num_tokens": 2183144.0,
994
+ "eval_runtime": 88.0652,
995
+ "eval_samples_per_second": 16.124,
996
+ "eval_steps_per_second": 2.021,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.6236911740154027,
1001
+ "epoch": 2.3138201569100785,
1002
+ "grad_norm": 0.7812452912330627,
1003
+ "learning_rate": 0.00035693211783557416,
1004
+ "loss": 0.5696200847625732,
1005
+ "mean_token_accuracy": 0.82991396561265,
1006
+ "num_tokens": 2228098.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3138201569100785,
1011
+ "eval_entropy": 0.6327033426319615,
1012
+ "eval_loss": 0.6742984652519226,
1013
+ "eval_mean_token_accuracy": 0.8148086091105857,
1014
+ "eval_num_tokens": 2228098.0,
1015
+ "eval_runtime": 88.1711,
1016
+ "eval_samples_per_second": 16.105,
1017
+ "eval_steps_per_second": 2.019,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.6218426413834095,
1022
+ "epoch": 2.3621001810500903,
1023
+ "grad_norm": 0.8719790577888489,
1024
+ "learning_rate": 0.0003555100848215035,
1025
+ "loss": 0.5582651615142822,
1026
+ "mean_token_accuracy": 0.8323961742222309,
1027
+ "num_tokens": 2273062.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.3621001810500903,
1032
+ "eval_entropy": 0.6338012375858393,
1033
+ "eval_loss": 0.6685822606086731,
1034
+ "eval_mean_token_accuracy": 0.8152180300669724,
1035
+ "eval_num_tokens": 2273062.0,
1036
+ "eval_runtime": 88.101,
1037
+ "eval_samples_per_second": 16.118,
1038
+ "eval_steps_per_second": 2.02,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.6050681818276644,
1043
+ "epoch": 2.4103802051901027,
1044
+ "grad_norm": 0.9151566028594971,
1045
+ "learning_rate": 0.00035404068900628076,
1046
+ "loss": 0.559452486038208,
1047
+ "mean_token_accuracy": 0.8310242861509323,
1048
+ "num_tokens": 2319305.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.4103802051901027,
1053
+ "eval_entropy": 0.6634890969549672,
1054
+ "eval_loss": 0.6692465543746948,
1055
+ "eval_mean_token_accuracy": 0.8134353673190213,
1056
+ "eval_num_tokens": 2319305.0,
1057
+ "eval_runtime": 87.7711,
1058
+ "eval_samples_per_second": 16.178,
1059
+ "eval_steps_per_second": 2.028,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.6169968567788601,
1064
+ "epoch": 2.4586602293301145,
1065
+ "grad_norm": 0.7081006765365601,
1066
+ "learning_rate": 0.0003525243462115402,
1067
+ "loss": 0.5589694023132324,
1068
+ "mean_token_accuracy": 0.8357502184808254,
1069
+ "num_tokens": 2368592.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.4586602293301145,
1074
+ "eval_entropy": 0.6126393578695447,
1075
+ "eval_loss": 0.6706712245941162,
1076
+ "eval_mean_token_accuracy": 0.8155577500884452,
1077
+ "eval_num_tokens": 2368592.0,
1078
+ "eval_runtime": 88.1407,
1079
+ "eval_samples_per_second": 16.111,
1080
+ "eval_steps_per_second": 2.019,
1081
+ "step": 1020
1082
+ }
1083
+ ],
1084
+ "logging_steps": 20,
1085
+ "max_steps": 4150,
1086
+ "num_input_tokens_seen": 0,
1087
+ "num_train_epochs": 10,
1088
+ "save_steps": 20,
1089
+ "stateful_callbacks": {
1090
+ "TrainerControl": {
1091
+ "args": {
1092
+ "should_epoch_stop": false,
1093
+ "should_evaluate": false,
1094
+ "should_log": false,
1095
+ "should_save": true,
1096
+ "should_training_stop": false
1097
+ },
1098
+ "attributes": {}
1099
+ }
1100
+ },
1101
+ "total_flos": 1.0007168784646349e+17,
1102
+ "train_batch_size": 4,
1103
+ "trial_name": null,
1104
+ "trial_params": null
1105
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.03828026524568501,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "v_proj",
34
+ "up_proj",
35
+ "down_proj",
36
+ "o_proj",
37
+ "q_proj",
38
+ "k_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/trainer_state.json ADDED
@@ -0,0 +1,1126 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.506940253470127,
6
+ "eval_steps": 20,
7
+ "global_step": 1040,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 2.172999547421932,
14
+ "epoch": 0.04828002414001207,
15
+ "grad_norm": 2.5071208477020264,
16
+ "learning_rate": 1.7227585487708594e-05,
17
+ "loss": 1.9347333908081055,
18
+ "mean_token_accuracy": 0.6086816020309925,
19
+ "num_tokens": 47778.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.04828002414001207,
24
+ "eval_entropy": 1.6909754791956269,
25
+ "eval_loss": 1.4728385210037231,
26
+ "eval_mean_token_accuracy": 0.6773593797442619,
27
+ "eval_num_tokens": 47778.0,
28
+ "eval_runtime": 90.1265,
29
+ "eval_samples_per_second": 15.756,
30
+ "eval_steps_per_second": 1.975,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.2014626950025558,
35
+ "epoch": 0.09656004828002414,
36
+ "grad_norm": 1.5074543952941895,
37
+ "learning_rate": 3.5361886001086065e-05,
38
+ "loss": 1.0867146492004394,
39
+ "mean_token_accuracy": 0.7285927519202232,
40
+ "num_tokens": 95981.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.09656004828002414,
45
+ "eval_entropy": 1.0304168352250302,
46
+ "eval_loss": 0.9211422204971313,
47
+ "eval_mean_token_accuracy": 0.7613122848312507,
48
+ "eval_num_tokens": 95981.0,
49
+ "eval_runtime": 88.6338,
50
+ "eval_samples_per_second": 16.021,
51
+ "eval_steps_per_second": 2.008,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.9454198732972146,
56
+ "epoch": 0.14484007242003621,
57
+ "grad_norm": 1.7578108310699463,
58
+ "learning_rate": 5.349618651446354e-05,
59
+ "loss": 0.8447297096252442,
60
+ "mean_token_accuracy": 0.7723424732685089,
61
+ "num_tokens": 142784.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.14484007242003621,
66
+ "eval_entropy": 0.8647834474451086,
67
+ "eval_loss": 0.8240737318992615,
68
+ "eval_mean_token_accuracy": 0.7821227014064789,
69
+ "eval_num_tokens": 142784.0,
70
+ "eval_runtime": 88.0135,
71
+ "eval_samples_per_second": 16.134,
72
+ "eval_steps_per_second": 2.022,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.8963550612330436,
77
+ "epoch": 0.19312009656004828,
78
+ "grad_norm": 1.1464931964874268,
79
+ "learning_rate": 7.1630487027841e-05,
80
+ "loss": 0.7979836463928223,
81
+ "mean_token_accuracy": 0.7822489373385906,
82
+ "num_tokens": 187338.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.19312009656004828,
87
+ "eval_entropy": 0.8663296133614657,
88
+ "eval_loss": 0.7752988338470459,
89
+ "eval_mean_token_accuracy": 0.7885957012015782,
90
+ "eval_num_tokens": 187338.0,
91
+ "eval_runtime": 88.284,
92
+ "eval_samples_per_second": 16.084,
93
+ "eval_steps_per_second": 2.016,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.8265103206038475,
98
+ "epoch": 0.24140012070006034,
99
+ "grad_norm": 1.0106662511825562,
100
+ "learning_rate": 8.976478754121848e-05,
101
+ "loss": 0.7297060012817382,
102
+ "mean_token_accuracy": 0.7955138415098191,
103
+ "num_tokens": 236038.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24140012070006034,
108
+ "eval_entropy": 0.8122169536151244,
109
+ "eval_loss": 0.754906177520752,
110
+ "eval_mean_token_accuracy": 0.7917155255092664,
111
+ "eval_num_tokens": 236038.0,
112
+ "eval_runtime": 87.5754,
113
+ "eval_samples_per_second": 16.215,
114
+ "eval_steps_per_second": 2.033,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.8098392844200134,
119
+ "epoch": 0.28968014484007243,
120
+ "grad_norm": 1.2238041162490845,
121
+ "learning_rate": 0.00010789908805459595,
122
+ "loss": 0.7201489448547364,
123
+ "mean_token_accuracy": 0.7975016921758652,
124
+ "num_tokens": 282918.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.28968014484007243,
129
+ "eval_entropy": 0.8418726462326692,
130
+ "eval_loss": 0.7393178939819336,
131
+ "eval_mean_token_accuracy": 0.7936771585700217,
132
+ "eval_num_tokens": 282918.0,
133
+ "eval_runtime": 87.1597,
134
+ "eval_samples_per_second": 16.292,
135
+ "eval_steps_per_second": 2.042,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.8080591082572937,
140
+ "epoch": 0.33796016898008446,
141
+ "grad_norm": 1.011591911315918,
142
+ "learning_rate": 0.00012603338856797343,
143
+ "loss": 0.7147697925567627,
144
+ "mean_token_accuracy": 0.7947175532579422,
145
+ "num_tokens": 331305.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.33796016898008446,
150
+ "eval_entropy": 0.7894941431083037,
151
+ "eval_loss": 0.720219075679779,
152
+ "eval_mean_token_accuracy": 0.801899236239744,
153
+ "eval_num_tokens": 331305.0,
154
+ "eval_runtime": 87.1493,
155
+ "eval_samples_per_second": 16.294,
156
+ "eval_steps_per_second": 2.042,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.7957205109298229,
161
+ "epoch": 0.38624019312009655,
162
+ "grad_norm": 0.8481830954551697,
163
+ "learning_rate": 0.00014416768908135088,
164
+ "loss": 0.7050027847290039,
165
+ "mean_token_accuracy": 0.7995368830859662,
166
+ "num_tokens": 375876.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.38624019312009655,
171
+ "eval_entropy": 0.7746140591883928,
172
+ "eval_loss": 0.7175475358963013,
173
+ "eval_mean_token_accuracy": 0.7994848955213354,
174
+ "eval_num_tokens": 375876.0,
175
+ "eval_runtime": 86.655,
176
+ "eval_samples_per_second": 16.387,
177
+ "eval_steps_per_second": 2.054,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.779565304517746,
182
+ "epoch": 0.43452021726010864,
183
+ "grad_norm": 1.032624363899231,
184
+ "learning_rate": 0.00016230198959472835,
185
+ "loss": 0.6978332042694092,
186
+ "mean_token_accuracy": 0.8015618294477462,
187
+ "num_tokens": 423913.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.43452021726010864,
192
+ "eval_entropy": 0.7455583871080634,
193
+ "eval_loss": 0.7118850350379944,
194
+ "eval_mean_token_accuracy": 0.7999682195400923,
195
+ "eval_num_tokens": 423913.0,
196
+ "eval_runtime": 86.7842,
197
+ "eval_samples_per_second": 16.362,
198
+ "eval_steps_per_second": 2.051,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.7810172818601131,
203
+ "epoch": 0.4828002414001207,
204
+ "grad_norm": 0.9276663064956665,
205
+ "learning_rate": 0.00018043629010810582,
206
+ "loss": 0.6937861442565918,
207
+ "mean_token_accuracy": 0.802893053740263,
208
+ "num_tokens": 472395.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.4828002414001207,
213
+ "eval_entropy": 0.7650193290764027,
214
+ "eval_loss": 0.7020567059516907,
215
+ "eval_mean_token_accuracy": 0.8045302153973097,
216
+ "eval_num_tokens": 472395.0,
217
+ "eval_runtime": 88.326,
218
+ "eval_samples_per_second": 16.077,
219
+ "eval_steps_per_second": 2.015,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.7862060204148292,
224
+ "epoch": 0.5310802655401328,
225
+ "grad_norm": 0.8195295929908752,
226
+ "learning_rate": 0.0001985705906214833,
227
+ "loss": 0.6878782272338867,
228
+ "mean_token_accuracy": 0.8027168907225132,
229
+ "num_tokens": 517049.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.5310802655401328,
234
+ "eval_entropy": 0.7656565583154057,
235
+ "eval_loss": 0.6973932981491089,
236
+ "eval_mean_token_accuracy": 0.8047967278555538,
237
+ "eval_num_tokens": 517049.0,
238
+ "eval_runtime": 87.7698,
239
+ "eval_samples_per_second": 16.179,
240
+ "eval_steps_per_second": 2.028,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.7611534461379051,
245
+ "epoch": 0.5793602896801449,
246
+ "grad_norm": 0.7499712109565735,
247
+ "learning_rate": 0.00021670489113486077,
248
+ "loss": 0.681937837600708,
249
+ "mean_token_accuracy": 0.8041978515684605,
250
+ "num_tokens": 562836.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.5793602896801449,
255
+ "eval_entropy": 0.7599442380197933,
256
+ "eval_loss": 0.7002046704292297,
257
+ "eval_mean_token_accuracy": 0.8051861517884759,
258
+ "eval_num_tokens": 562836.0,
259
+ "eval_runtime": 87.9239,
260
+ "eval_samples_per_second": 16.15,
261
+ "eval_steps_per_second": 2.024,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.7800783462822437,
266
+ "epoch": 0.627640313820157,
267
+ "grad_norm": 0.8645269274711609,
268
+ "learning_rate": 0.00023483919164823825,
269
+ "loss": 0.6954500675201416,
270
+ "mean_token_accuracy": 0.8031484372913837,
271
+ "num_tokens": 606612.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.627640313820157,
276
+ "eval_entropy": 0.7498679090751691,
277
+ "eval_loss": 0.7046149373054504,
278
+ "eval_mean_token_accuracy": 0.802388215667746,
279
+ "eval_num_tokens": 606612.0,
280
+ "eval_runtime": 88.2468,
281
+ "eval_samples_per_second": 16.091,
282
+ "eval_steps_per_second": 2.017,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.7586533665657044,
287
+ "epoch": 0.6759203379601689,
288
+ "grad_norm": 0.741596519947052,
289
+ "learning_rate": 0.0002529734921616157,
290
+ "loss": 0.690476369857788,
291
+ "mean_token_accuracy": 0.8046923100948333,
292
+ "num_tokens": 655920.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6759203379601689,
297
+ "eval_entropy": 0.7714274066887544,
298
+ "eval_loss": 0.6992160677909851,
299
+ "eval_mean_token_accuracy": 0.8042646482419432,
300
+ "eval_num_tokens": 655920.0,
301
+ "eval_runtime": 88.1754,
302
+ "eval_samples_per_second": 16.104,
303
+ "eval_steps_per_second": 2.019,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.7585848286747933,
308
+ "epoch": 0.724200362100181,
309
+ "grad_norm": 1.0778124332427979,
310
+ "learning_rate": 0.00027110779267499317,
311
+ "loss": 0.6790409564971924,
312
+ "mean_token_accuracy": 0.8060351841151714,
313
+ "num_tokens": 702638.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.724200362100181,
318
+ "eval_entropy": 0.7380402811457601,
319
+ "eval_loss": 0.7054820656776428,
320
+ "eval_mean_token_accuracy": 0.8003534738267406,
321
+ "eval_num_tokens": 702638.0,
322
+ "eval_runtime": 88.238,
323
+ "eval_samples_per_second": 16.093,
324
+ "eval_steps_per_second": 2.017,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.7696981988847256,
329
+ "epoch": 0.7724803862401931,
330
+ "grad_norm": 0.8408658504486084,
331
+ "learning_rate": 0.00028924209318837064,
332
+ "loss": 0.7070387363433838,
333
+ "mean_token_accuracy": 0.8035947173833847,
334
+ "num_tokens": 749377.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.7724803862401931,
339
+ "eval_entropy": 0.7897409023193831,
340
+ "eval_loss": 0.7015668749809265,
341
+ "eval_mean_token_accuracy": 0.805306687113944,
342
+ "eval_num_tokens": 749377.0,
343
+ "eval_runtime": 87.9395,
344
+ "eval_samples_per_second": 16.147,
345
+ "eval_steps_per_second": 2.024,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.762337189912796,
350
+ "epoch": 0.8207604103802052,
351
+ "grad_norm": 0.9843310117721558,
352
+ "learning_rate": 0.0003073763937017481,
353
+ "loss": 0.7021088123321533,
354
+ "mean_token_accuracy": 0.8029616877436638,
355
+ "num_tokens": 798874.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8207604103802052,
360
+ "eval_entropy": 0.7210077121016685,
361
+ "eval_loss": 0.7082746624946594,
362
+ "eval_mean_token_accuracy": 0.8049283824609906,
363
+ "eval_num_tokens": 798874.0,
364
+ "eval_runtime": 88.1706,
365
+ "eval_samples_per_second": 16.105,
366
+ "eval_steps_per_second": 2.019,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.7879025422036647,
371
+ "epoch": 0.8690404345202173,
372
+ "grad_norm": 0.9713814854621887,
373
+ "learning_rate": 0.0003255106942151256,
374
+ "loss": 0.714624547958374,
375
+ "mean_token_accuracy": 0.79987031519413,
376
+ "num_tokens": 840825.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8690404345202173,
381
+ "eval_entropy": 0.7866930489459735,
382
+ "eval_loss": 0.7145519852638245,
383
+ "eval_mean_token_accuracy": 0.8015657988157165,
384
+ "eval_num_tokens": 840825.0,
385
+ "eval_runtime": 88.3746,
386
+ "eval_samples_per_second": 16.068,
387
+ "eval_steps_per_second": 2.014,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.7844726584851742,
392
+ "epoch": 0.9173204586602294,
393
+ "grad_norm": 1.2028311491012573,
394
+ "learning_rate": 0.00034364499472850306,
395
+ "loss": 0.7040786266326904,
396
+ "mean_token_accuracy": 0.7981885217130185,
397
+ "num_tokens": 882885.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9173204586602294,
402
+ "eval_entropy": 0.7767668339643585,
403
+ "eval_loss": 0.7073574066162109,
404
+ "eval_mean_token_accuracy": 0.8040851833445303,
405
+ "eval_num_tokens": 882885.0,
406
+ "eval_runtime": 88.3828,
407
+ "eval_samples_per_second": 16.066,
408
+ "eval_steps_per_second": 2.014,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.7699790760874748,
413
+ "epoch": 0.9656004828002414,
414
+ "grad_norm": 0.9397398233413696,
415
+ "learning_rate": 0.00036177929524188054,
416
+ "loss": 0.7040313720703125,
417
+ "mean_token_accuracy": 0.8034727744758129,
418
+ "num_tokens": 927456.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9656004828002414,
423
+ "eval_entropy": 0.7749103545472863,
424
+ "eval_loss": 0.7093513011932373,
425
+ "eval_mean_token_accuracy": 0.8019482901926791,
426
+ "eval_num_tokens": 927456.0,
427
+ "eval_runtime": 86.2112,
428
+ "eval_samples_per_second": 16.471,
429
+ "eval_steps_per_second": 2.065,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.7672389172888422,
434
+ "epoch": 1.012070006035003,
435
+ "grad_norm": 0.99210524559021,
436
+ "learning_rate": 0.00037628567078152296,
437
+ "loss": 0.6913118362426758,
438
+ "mean_token_accuracy": 0.8042210944287189,
439
+ "num_tokens": 972662.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.012070006035003,
444
+ "eval_entropy": 0.7294771229283193,
445
+ "eval_loss": 0.7171520590782166,
446
+ "eval_mean_token_accuracy": 0.8015128585059991,
447
+ "eval_num_tokens": 972662.0,
448
+ "eval_runtime": 88.037,
449
+ "eval_samples_per_second": 16.13,
450
+ "eval_steps_per_second": 2.022,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.7143943183124065,
455
+ "epoch": 1.060350030175015,
456
+ "grad_norm": 1.087638258934021,
457
+ "learning_rate": 0.0003762484015601069,
458
+ "loss": 0.642704439163208,
459
+ "mean_token_accuracy": 0.8124966286122799,
460
+ "num_tokens": 1022525.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.060350030175015,
465
+ "eval_entropy": 0.743511886744017,
466
+ "eval_loss": 0.7116662859916687,
467
+ "eval_mean_token_accuracy": 0.8037210672758939,
468
+ "eval_num_tokens": 1022525.0,
469
+ "eval_runtime": 87.9931,
470
+ "eval_samples_per_second": 16.138,
471
+ "eval_steps_per_second": 2.023,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.7237346574664116,
476
+ "epoch": 1.1086300543150271,
477
+ "grad_norm": 0.9365553259849548,
478
+ "learning_rate": 0.0003761579008392123,
479
+ "loss": 0.652644681930542,
480
+ "mean_token_accuracy": 0.8107496216893196,
481
+ "num_tokens": 1069194.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1086300543150271,
486
+ "eval_entropy": 0.7372823839776972,
487
+ "eval_loss": 0.7238303422927856,
488
+ "eval_mean_token_accuracy": 0.8026230124275336,
489
+ "eval_num_tokens": 1069194.0,
490
+ "eval_runtime": 87.6778,
491
+ "eval_samples_per_second": 16.196,
492
+ "eval_steps_per_second": 2.03,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.7202137842774391,
497
+ "epoch": 1.1569100784550392,
498
+ "grad_norm": 0.7998443841934204,
499
+ "learning_rate": 0.0003760141942294726,
500
+ "loss": 0.6485219478607178,
501
+ "mean_token_accuracy": 0.8092714451253414,
502
+ "num_tokens": 1117570.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1569100784550392,
507
+ "eval_entropy": 0.71547647942318,
508
+ "eval_loss": 0.7124277949333191,
509
+ "eval_mean_token_accuracy": 0.8013174219077892,
510
+ "eval_num_tokens": 1117570.0,
511
+ "eval_runtime": 87.4387,
512
+ "eval_samples_per_second": 16.24,
513
+ "eval_steps_per_second": 2.036,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.7270819500088692,
518
+ "epoch": 1.2051901025950513,
519
+ "grad_norm": 0.9470787644386292,
520
+ "learning_rate": 0.00037581732239815854,
521
+ "loss": 0.66352219581604,
522
+ "mean_token_accuracy": 0.8102593503892421,
523
+ "num_tokens": 1161717.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.2051901025950513,
528
+ "eval_entropy": 0.7791565897759427,
529
+ "eval_loss": 0.7179726958274841,
530
+ "eval_mean_token_accuracy": 0.7948898328154275,
531
+ "eval_num_tokens": 1161717.0,
532
+ "eval_runtime": 87.9927,
533
+ "eval_samples_per_second": 16.138,
534
+ "eval_steps_per_second": 2.023,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.7320936664938926,
539
+ "epoch": 1.2534701267350634,
540
+ "grad_norm": 1.0220379829406738,
541
+ "learning_rate": 0.0003755673410576695,
542
+ "loss": 0.6581549644470215,
543
+ "mean_token_accuracy": 0.8098709337413311,
544
+ "num_tokens": 1209390.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.2534701267350634,
549
+ "eval_entropy": 0.7710678225822662,
550
+ "eval_loss": 0.708797812461853,
551
+ "eval_mean_token_accuracy": 0.8039572778042783,
552
+ "eval_num_tokens": 1209390.0,
553
+ "eval_runtime": 88.1193,
554
+ "eval_samples_per_second": 16.115,
555
+ "eval_steps_per_second": 2.02,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.7281674664467573,
560
+ "epoch": 1.3017501508750755,
561
+ "grad_norm": 0.8775188326835632,
562
+ "learning_rate": 0.000375264320949768,
563
+ "loss": 0.6631278991699219,
564
+ "mean_token_accuracy": 0.8101261422038079,
565
+ "num_tokens": 1256611.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3017501508750755,
570
+ "eval_entropy": 0.7199799800187014,
571
+ "eval_loss": 0.6981531381607056,
572
+ "eval_mean_token_accuracy": 0.8071265254127845,
573
+ "eval_num_tokens": 1256611.0,
574
+ "eval_runtime": 88.165,
575
+ "eval_samples_per_second": 16.106,
576
+ "eval_steps_per_second": 2.019,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.722860773652792,
581
+ "epoch": 1.3500301750150876,
582
+ "grad_norm": 0.8949288129806519,
583
+ "learning_rate": 0.00037490834782556,
584
+ "loss": 0.6639598846435547,
585
+ "mean_token_accuracy": 0.8105649061501026,
586
+ "num_tokens": 1302325.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3500301750150876,
591
+ "eval_entropy": 0.7134760032209118,
592
+ "eval_loss": 0.7029948234558105,
593
+ "eval_mean_token_accuracy": 0.8062985880991046,
594
+ "eval_num_tokens": 1302325.0,
595
+ "eval_runtime": 88.1286,
596
+ "eval_samples_per_second": 16.113,
597
+ "eval_steps_per_second": 2.02,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.7299126699566841,
602
+ "epoch": 1.3983101991550995,
603
+ "grad_norm": 1.0323039293289185,
604
+ "learning_rate": 0.0003744995224212291,
605
+ "loss": 0.6535763263702392,
606
+ "mean_token_accuracy": 0.8120224848389626,
607
+ "num_tokens": 1343844.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.3983101991550995,
612
+ "eval_entropy": 0.7169701182440426,
613
+ "eval_loss": 0.7067857980728149,
614
+ "eval_mean_token_accuracy": 0.8040432936689826,
615
+ "eval_num_tokens": 1343844.0,
616
+ "eval_runtime": 88.2252,
617
+ "eval_samples_per_second": 16.095,
618
+ "eval_steps_per_second": 2.018,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.7281742256134749,
623
+ "epoch": 1.4465902232951118,
624
+ "grad_norm": 1.3370424509048462,
625
+ "learning_rate": 0.00037403796042952863,
626
+ "loss": 0.6637272357940673,
627
+ "mean_token_accuracy": 0.8118261776864528,
628
+ "num_tokens": 1389471.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4465902232951118,
633
+ "eval_entropy": 0.7378843212395572,
634
+ "eval_loss": 0.7038390040397644,
635
+ "eval_mean_token_accuracy": 0.8059229371922739,
636
+ "eval_num_tokens": 1389471.0,
637
+ "eval_runtime": 87.9301,
638
+ "eval_samples_per_second": 16.149,
639
+ "eval_steps_per_second": 2.024,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.7388614989817143,
644
+ "epoch": 1.4948702474351236,
645
+ "grad_norm": 0.8464171290397644,
646
+ "learning_rate": 0.00037352379246704257,
647
+ "loss": 0.6604740142822265,
648
+ "mean_token_accuracy": 0.8102974124252796,
649
+ "num_tokens": 1435975.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.4948702474351236,
654
+ "eval_entropy": 0.7126847046814607,
655
+ "eval_loss": 0.6950703263282776,
656
+ "eval_mean_token_accuracy": 0.8085885392815879,
657
+ "eval_num_tokens": 1435975.0,
658
+ "eval_runtime": 88.2732,
659
+ "eval_samples_per_second": 16.086,
660
+ "eval_steps_per_second": 2.016,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.7258916139602661,
665
+ "epoch": 1.5431502715751357,
666
+ "grad_norm": 1.1780085563659668,
667
+ "learning_rate": 0.0003729571640372223,
668
+ "loss": 0.6529891014099121,
669
+ "mean_token_accuracy": 0.8085981778800487,
670
+ "num_tokens": 1482035.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.5431502715751357,
675
+ "eval_entropy": 0.7096509638797032,
676
+ "eval_loss": 0.690051257610321,
677
+ "eval_mean_token_accuracy": 0.8100520241796301,
678
+ "eval_num_tokens": 1482035.0,
679
+ "eval_runtime": 88.3153,
680
+ "eval_samples_per_second": 16.079,
681
+ "eval_steps_per_second": 2.016,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.7151173129677773,
686
+ "epoch": 1.5914302957151478,
687
+ "grad_norm": 1.0357667207717896,
688
+ "learning_rate": 0.0003723382354892108,
689
+ "loss": 0.6451474189758301,
690
+ "mean_token_accuracy": 0.8136902332305909,
691
+ "num_tokens": 1528119.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.5914302957151478,
696
+ "eval_entropy": 0.7060252652409371,
697
+ "eval_loss": 0.6899483799934387,
698
+ "eval_mean_token_accuracy": 0.810834194502134,
699
+ "eval_num_tokens": 1528119.0,
700
+ "eval_runtime": 87.8848,
701
+ "eval_samples_per_second": 16.158,
702
+ "eval_steps_per_second": 2.025,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.7141377851366997,
707
+ "epoch": 1.63971031985516,
708
+ "grad_norm": 0.8391667008399963,
709
+ "learning_rate": 0.000371667181972466,
710
+ "loss": 0.6481579780578614,
711
+ "mean_token_accuracy": 0.8159015104174614,
712
+ "num_tokens": 1573661.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.63971031985516,
717
+ "eval_entropy": 0.7598993319473909,
718
+ "eval_loss": 0.681204080581665,
719
+ "eval_mean_token_accuracy": 0.8100194586126992,
720
+ "eval_num_tokens": 1573661.0,
721
+ "eval_runtime": 88.2415,
722
+ "eval_samples_per_second": 16.092,
723
+ "eval_steps_per_second": 2.017,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.7235980287194252,
728
+ "epoch": 1.687990343995172,
729
+ "grad_norm": 0.8197196125984192,
730
+ "learning_rate": 0.00037094419338719537,
731
+ "loss": 0.6582849025726318,
732
+ "mean_token_accuracy": 0.8114834539592266,
733
+ "num_tokens": 1621714.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.687990343995172,
738
+ "eval_entropy": 0.7166854568411795,
739
+ "eval_loss": 0.679177463054657,
740
+ "eval_mean_token_accuracy": 0.8119039877077167,
741
+ "eval_num_tokens": 1621714.0,
742
+ "eval_runtime": 88.0957,
743
+ "eval_samples_per_second": 16.119,
744
+ "eval_steps_per_second": 2.021,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.7278190158307553,
749
+ "epoch": 1.736270368135184,
750
+ "grad_norm": 0.7820602655410767,
751
+ "learning_rate": 0.0003701694743306164,
752
+ "loss": 0.6633153438568116,
753
+ "mean_token_accuracy": 0.8123249113559723,
754
+ "num_tokens": 1672118.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.736270368135184,
759
+ "eval_entropy": 0.6980850314826108,
760
+ "eval_loss": 0.6775653958320618,
761
+ "eval_mean_token_accuracy": 0.8116872906684875,
762
+ "eval_num_tokens": 1672118.0,
763
+ "eval_runtime": 87.9795,
764
+ "eval_samples_per_second": 16.14,
765
+ "eval_steps_per_second": 2.023,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.7149631556123495,
770
+ "epoch": 1.7845503922751962,
771
+ "grad_norm": 1.0139760971069336,
772
+ "learning_rate": 0.00036934324403905795,
773
+ "loss": 0.6468976020812989,
774
+ "mean_token_accuracy": 0.8117561548948288,
775
+ "num_tokens": 1720465.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.7845503922751962,
780
+ "eval_entropy": 0.69009849094273,
781
+ "eval_loss": 0.6731093525886536,
782
+ "eval_mean_token_accuracy": 0.81281964196248,
783
+ "eval_num_tokens": 1720465.0,
784
+ "eval_runtime": 87.7412,
785
+ "eval_samples_per_second": 16.184,
786
+ "eval_steps_per_second": 2.029,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.7226049453020096,
791
+ "epoch": 1.832830416415208,
792
+ "grad_norm": 1.5040546655654907,
793
+ "learning_rate": 0.0003684657363259193,
794
+ "loss": 0.6385328769683838,
795
+ "mean_token_accuracy": 0.8132639616727829,
796
+ "num_tokens": 1764440.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.832830416415208,
801
+ "eval_entropy": 0.7125169618076153,
802
+ "eval_loss": 0.6864680647850037,
803
+ "eval_mean_token_accuracy": 0.8049418089095126,
804
+ "eval_num_tokens": 1764440.0,
805
+ "eval_runtime": 87.9429,
806
+ "eval_samples_per_second": 16.147,
807
+ "eval_steps_per_second": 2.024,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.707821810990572,
812
+ "epoch": 1.8811104405552204,
813
+ "grad_norm": 0.8604223132133484,
814
+ "learning_rate": 0.00036753719951550327,
815
+ "loss": 0.6397049427032471,
816
+ "mean_token_accuracy": 0.8137877315282822,
817
+ "num_tokens": 1810828.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.8811104405552204,
822
+ "eval_entropy": 0.6492102551326323,
823
+ "eval_loss": 0.6743726134300232,
824
+ "eval_mean_token_accuracy": 0.8130356572317273,
825
+ "eval_num_tokens": 1810828.0,
826
+ "eval_runtime": 87.8684,
827
+ "eval_samples_per_second": 16.161,
828
+ "eval_steps_per_second": 2.026,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.6800355084240437,
833
+ "epoch": 1.9293904646952322,
834
+ "grad_norm": 0.7270982265472412,
835
+ "learning_rate": 0.00036655789637274377,
836
+ "loss": 0.6267755031585693,
837
+ "mean_token_accuracy": 0.8180945307016373,
838
+ "num_tokens": 1857600.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9293904646952322,
843
+ "eval_entropy": 0.6727538651294922,
844
+ "eval_loss": 0.6696028709411621,
845
+ "eval_mean_token_accuracy": 0.8116443090224534,
846
+ "eval_num_tokens": 1857600.0,
847
+ "eval_runtime": 88.2449,
848
+ "eval_samples_per_second": 16.092,
849
+ "eval_steps_per_second": 2.017,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.6746691606938839,
854
+ "epoch": 1.9776704888352445,
855
+ "grad_norm": 0.9771540760993958,
856
+ "learning_rate": 0.0003655281040288461,
857
+ "loss": 0.6326992988586426,
858
+ "mean_token_accuracy": 0.8179976396262646,
859
+ "num_tokens": 1901998.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 1.9776704888352445,
864
+ "eval_entropy": 0.6636319716324967,
865
+ "eval_loss": 0.6714363098144531,
866
+ "eval_mean_token_accuracy": 0.8126635943235976,
867
+ "eval_num_tokens": 1901998.0,
868
+ "eval_runtime": 87.7666,
869
+ "eval_samples_per_second": 16.179,
870
+ "eval_steps_per_second": 2.028,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.6319125778489298,
875
+ "epoch": 2.024140012070006,
876
+ "grad_norm": 0.8380444645881653,
877
+ "learning_rate": 0.0003644481139028622,
878
+ "loss": 0.5780903816223144,
879
+ "mean_token_accuracy": 0.8250188401767186,
880
+ "num_tokens": 1947553.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.024140012070006,
885
+ "eval_entropy": 0.6627151989535,
886
+ "eval_loss": 0.6774935722351074,
887
+ "eval_mean_token_accuracy": 0.8140510819601209,
888
+ "eval_num_tokens": 1947553.0,
889
+ "eval_runtime": 87.8287,
890
+ "eval_samples_per_second": 16.168,
891
+ "eval_steps_per_second": 2.027,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.6059147048741579,
896
+ "epoch": 2.0724200362100182,
897
+ "grad_norm": 0.9745152592658997,
898
+ "learning_rate": 0.0003633182316192229,
899
+ "loss": 0.546702527999878,
900
+ "mean_token_accuracy": 0.8358186542987823,
901
+ "num_tokens": 1995234.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.0724200362100182,
906
+ "eval_entropy": 0.6620710384979677,
907
+ "eval_loss": 0.6769291162490845,
908
+ "eval_mean_token_accuracy": 0.8135082835561773,
909
+ "eval_num_tokens": 1995234.0,
910
+ "eval_runtime": 88.0008,
911
+ "eval_samples_per_second": 16.136,
912
+ "eval_steps_per_second": 2.023,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.6076117843389511,
917
+ "epoch": 2.12070006035003,
918
+ "grad_norm": 0.6914380192756653,
919
+ "learning_rate": 0.0003621387769212491,
920
+ "loss": 0.5324535846710206,
921
+ "mean_token_accuracy": 0.8366568259894848,
922
+ "num_tokens": 2040639.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.12070006035003,
927
+ "eval_entropy": 0.6592224213514435,
928
+ "eval_loss": 0.670383632183075,
929
+ "eval_mean_token_accuracy": 0.8162150088320957,
930
+ "eval_num_tokens": 2040639.0,
931
+ "eval_runtime": 88.324,
932
+ "eval_samples_per_second": 16.077,
933
+ "eval_steps_per_second": 2.015,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.6158072650432587,
938
+ "epoch": 2.1689800844900424,
939
+ "grad_norm": 0.9649198055267334,
940
+ "learning_rate": 0.0003609100835806688,
941
+ "loss": 0.5564568996429443,
942
+ "mean_token_accuracy": 0.8329654954373836,
943
+ "num_tokens": 2090604.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.1689800844900424,
948
+ "eval_entropy": 0.608588819758276,
949
+ "eval_loss": 0.6745102405548096,
950
+ "eval_mean_token_accuracy": 0.8155080542135774,
951
+ "eval_num_tokens": 2090604.0,
952
+ "eval_runtime": 87.52,
953
+ "eval_samples_per_second": 16.225,
954
+ "eval_steps_per_second": 2.034,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.5938734326511621,
959
+ "epoch": 2.2172601086300543,
960
+ "grad_norm": 0.6915557384490967,
961
+ "learning_rate": 0.0003596324993031632,
962
+ "loss": 0.5351875782012939,
963
+ "mean_token_accuracy": 0.8385631121695042,
964
+ "num_tokens": 2142246.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2172601086300543,
969
+ "eval_entropy": 0.6251631205001574,
970
+ "eval_loss": 0.6664518117904663,
971
+ "eval_mean_token_accuracy": 0.81711940417129,
972
+ "eval_num_tokens": 2142246.0,
973
+ "eval_runtime": 87.1792,
974
+ "eval_samples_per_second": 16.288,
975
+ "eval_steps_per_second": 2.042,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.623173613846302,
980
+ "epoch": 2.2655401327700666,
981
+ "grad_norm": 0.8366610407829285,
982
+ "learning_rate": 0.00035830638562997063,
983
+ "loss": 0.5488213539123535,
984
+ "mean_token_accuracy": 0.8321483485400677,
985
+ "num_tokens": 2183144.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.2655401327700666,
990
+ "eval_entropy": 0.6237085649136747,
991
+ "eval_loss": 0.68075031042099,
992
+ "eval_mean_token_accuracy": 0.814671738429016,
993
+ "eval_num_tokens": 2183144.0,
994
+ "eval_runtime": 88.0652,
995
+ "eval_samples_per_second": 16.124,
996
+ "eval_steps_per_second": 2.021,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.6236911740154027,
1001
+ "epoch": 2.3138201569100785,
1002
+ "grad_norm": 0.7812452912330627,
1003
+ "learning_rate": 0.00035693211783557416,
1004
+ "loss": 0.5696200847625732,
1005
+ "mean_token_accuracy": 0.82991396561265,
1006
+ "num_tokens": 2228098.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3138201569100785,
1011
+ "eval_entropy": 0.6327033426319615,
1012
+ "eval_loss": 0.6742984652519226,
1013
+ "eval_mean_token_accuracy": 0.8148086091105857,
1014
+ "eval_num_tokens": 2228098.0,
1015
+ "eval_runtime": 88.1711,
1016
+ "eval_samples_per_second": 16.105,
1017
+ "eval_steps_per_second": 2.019,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.6218426413834095,
1022
+ "epoch": 2.3621001810500903,
1023
+ "grad_norm": 0.8719790577888489,
1024
+ "learning_rate": 0.0003555100848215035,
1025
+ "loss": 0.5582651615142822,
1026
+ "mean_token_accuracy": 0.8323961742222309,
1027
+ "num_tokens": 2273062.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.3621001810500903,
1032
+ "eval_entropy": 0.6338012375858393,
1033
+ "eval_loss": 0.6685822606086731,
1034
+ "eval_mean_token_accuracy": 0.8152180300669724,
1035
+ "eval_num_tokens": 2273062.0,
1036
+ "eval_runtime": 88.101,
1037
+ "eval_samples_per_second": 16.118,
1038
+ "eval_steps_per_second": 2.02,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.6050681818276644,
1043
+ "epoch": 2.4103802051901027,
1044
+ "grad_norm": 0.9151566028594971,
1045
+ "learning_rate": 0.00035404068900628076,
1046
+ "loss": 0.559452486038208,
1047
+ "mean_token_accuracy": 0.8310242861509323,
1048
+ "num_tokens": 2319305.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.4103802051901027,
1053
+ "eval_entropy": 0.6634890969549672,
1054
+ "eval_loss": 0.6692465543746948,
1055
+ "eval_mean_token_accuracy": 0.8134353673190213,
1056
+ "eval_num_tokens": 2319305.0,
1057
+ "eval_runtime": 87.7711,
1058
+ "eval_samples_per_second": 16.178,
1059
+ "eval_steps_per_second": 2.028,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.6169968567788601,
1064
+ "epoch": 2.4586602293301145,
1065
+ "grad_norm": 0.7081006765365601,
1066
+ "learning_rate": 0.0003525243462115402,
1067
+ "loss": 0.5589694023132324,
1068
+ "mean_token_accuracy": 0.8357502184808254,
1069
+ "num_tokens": 2368592.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.4586602293301145,
1074
+ "eval_entropy": 0.6126393578695447,
1075
+ "eval_loss": 0.6706712245941162,
1076
+ "eval_mean_token_accuracy": 0.8155577500884452,
1077
+ "eval_num_tokens": 2368592.0,
1078
+ "eval_runtime": 88.1407,
1079
+ "eval_samples_per_second": 16.111,
1080
+ "eval_steps_per_second": 2.019,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.6072248566895724,
1085
+ "epoch": 2.506940253470127,
1086
+ "grad_norm": 0.6999370455741882,
1087
+ "learning_rate": 0.0003509614855443561,
1088
+ "loss": 0.5500593185424805,
1089
+ "mean_token_accuracy": 0.8337264291942119,
1090
+ "num_tokens": 2412859.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.506940253470127,
1095
+ "eval_entropy": 0.6417417074187418,
1096
+ "eval_loss": 0.6651163697242737,
1097
+ "eval_mean_token_accuracy": 0.8159113908081912,
1098
+ "eval_num_tokens": 2412859.0,
1099
+ "eval_runtime": 87.4854,
1100
+ "eval_samples_per_second": 16.231,
1101
+ "eval_steps_per_second": 2.035,
1102
+ "step": 1040
1103
+ }
1104
+ ],
1105
+ "logging_steps": 20,
1106
+ "max_steps": 4150,
1107
+ "num_input_tokens_seen": 0,
1108
+ "num_train_epochs": 10,
1109
+ "save_steps": 20,
1110
+ "stateful_callbacks": {
1111
+ "TrainerControl": {
1112
+ "args": {
1113
+ "should_epoch_stop": false,
1114
+ "should_evaluate": false,
1115
+ "should_log": false,
1116
+ "should_save": true,
1117
+ "should_training_stop": false
1118
+ },
1119
+ "attributes": {}
1120
+ }
1121
+ },
1122
+ "total_flos": 1.0198081900655616e+17,
1123
+ "train_batch_size": 4,
1124
+ "trial_name": null,
1125
+ "trial_params": null
1126
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.03828026524568501,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "v_proj",
34
+ "up_proj",
35
+ "down_proj",
36
+ "o_proj",
37
+ "q_proj",
38
+ "k_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/trainer_state.json ADDED
@@ -0,0 +1,1147 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.5552202776101387,
6
+ "eval_steps": 20,
7
+ "global_step": 1060,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 2.172999547421932,
14
+ "epoch": 0.04828002414001207,
15
+ "grad_norm": 2.5071208477020264,
16
+ "learning_rate": 1.7227585487708594e-05,
17
+ "loss": 1.9347333908081055,
18
+ "mean_token_accuracy": 0.6086816020309925,
19
+ "num_tokens": 47778.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.04828002414001207,
24
+ "eval_entropy": 1.6909754791956269,
25
+ "eval_loss": 1.4728385210037231,
26
+ "eval_mean_token_accuracy": 0.6773593797442619,
27
+ "eval_num_tokens": 47778.0,
28
+ "eval_runtime": 90.1265,
29
+ "eval_samples_per_second": 15.756,
30
+ "eval_steps_per_second": 1.975,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.2014626950025558,
35
+ "epoch": 0.09656004828002414,
36
+ "grad_norm": 1.5074543952941895,
37
+ "learning_rate": 3.5361886001086065e-05,
38
+ "loss": 1.0867146492004394,
39
+ "mean_token_accuracy": 0.7285927519202232,
40
+ "num_tokens": 95981.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.09656004828002414,
45
+ "eval_entropy": 1.0304168352250302,
46
+ "eval_loss": 0.9211422204971313,
47
+ "eval_mean_token_accuracy": 0.7613122848312507,
48
+ "eval_num_tokens": 95981.0,
49
+ "eval_runtime": 88.6338,
50
+ "eval_samples_per_second": 16.021,
51
+ "eval_steps_per_second": 2.008,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.9454198732972146,
56
+ "epoch": 0.14484007242003621,
57
+ "grad_norm": 1.7578108310699463,
58
+ "learning_rate": 5.349618651446354e-05,
59
+ "loss": 0.8447297096252442,
60
+ "mean_token_accuracy": 0.7723424732685089,
61
+ "num_tokens": 142784.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.14484007242003621,
66
+ "eval_entropy": 0.8647834474451086,
67
+ "eval_loss": 0.8240737318992615,
68
+ "eval_mean_token_accuracy": 0.7821227014064789,
69
+ "eval_num_tokens": 142784.0,
70
+ "eval_runtime": 88.0135,
71
+ "eval_samples_per_second": 16.134,
72
+ "eval_steps_per_second": 2.022,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.8963550612330436,
77
+ "epoch": 0.19312009656004828,
78
+ "grad_norm": 1.1464931964874268,
79
+ "learning_rate": 7.1630487027841e-05,
80
+ "loss": 0.7979836463928223,
81
+ "mean_token_accuracy": 0.7822489373385906,
82
+ "num_tokens": 187338.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.19312009656004828,
87
+ "eval_entropy": 0.8663296133614657,
88
+ "eval_loss": 0.7752988338470459,
89
+ "eval_mean_token_accuracy": 0.7885957012015782,
90
+ "eval_num_tokens": 187338.0,
91
+ "eval_runtime": 88.284,
92
+ "eval_samples_per_second": 16.084,
93
+ "eval_steps_per_second": 2.016,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.8265103206038475,
98
+ "epoch": 0.24140012070006034,
99
+ "grad_norm": 1.0106662511825562,
100
+ "learning_rate": 8.976478754121848e-05,
101
+ "loss": 0.7297060012817382,
102
+ "mean_token_accuracy": 0.7955138415098191,
103
+ "num_tokens": 236038.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24140012070006034,
108
+ "eval_entropy": 0.8122169536151244,
109
+ "eval_loss": 0.754906177520752,
110
+ "eval_mean_token_accuracy": 0.7917155255092664,
111
+ "eval_num_tokens": 236038.0,
112
+ "eval_runtime": 87.5754,
113
+ "eval_samples_per_second": 16.215,
114
+ "eval_steps_per_second": 2.033,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.8098392844200134,
119
+ "epoch": 0.28968014484007243,
120
+ "grad_norm": 1.2238041162490845,
121
+ "learning_rate": 0.00010789908805459595,
122
+ "loss": 0.7201489448547364,
123
+ "mean_token_accuracy": 0.7975016921758652,
124
+ "num_tokens": 282918.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.28968014484007243,
129
+ "eval_entropy": 0.8418726462326692,
130
+ "eval_loss": 0.7393178939819336,
131
+ "eval_mean_token_accuracy": 0.7936771585700217,
132
+ "eval_num_tokens": 282918.0,
133
+ "eval_runtime": 87.1597,
134
+ "eval_samples_per_second": 16.292,
135
+ "eval_steps_per_second": 2.042,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.8080591082572937,
140
+ "epoch": 0.33796016898008446,
141
+ "grad_norm": 1.011591911315918,
142
+ "learning_rate": 0.00012603338856797343,
143
+ "loss": 0.7147697925567627,
144
+ "mean_token_accuracy": 0.7947175532579422,
145
+ "num_tokens": 331305.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.33796016898008446,
150
+ "eval_entropy": 0.7894941431083037,
151
+ "eval_loss": 0.720219075679779,
152
+ "eval_mean_token_accuracy": 0.801899236239744,
153
+ "eval_num_tokens": 331305.0,
154
+ "eval_runtime": 87.1493,
155
+ "eval_samples_per_second": 16.294,
156
+ "eval_steps_per_second": 2.042,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.7957205109298229,
161
+ "epoch": 0.38624019312009655,
162
+ "grad_norm": 0.8481830954551697,
163
+ "learning_rate": 0.00014416768908135088,
164
+ "loss": 0.7050027847290039,
165
+ "mean_token_accuracy": 0.7995368830859662,
166
+ "num_tokens": 375876.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.38624019312009655,
171
+ "eval_entropy": 0.7746140591883928,
172
+ "eval_loss": 0.7175475358963013,
173
+ "eval_mean_token_accuracy": 0.7994848955213354,
174
+ "eval_num_tokens": 375876.0,
175
+ "eval_runtime": 86.655,
176
+ "eval_samples_per_second": 16.387,
177
+ "eval_steps_per_second": 2.054,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.779565304517746,
182
+ "epoch": 0.43452021726010864,
183
+ "grad_norm": 1.032624363899231,
184
+ "learning_rate": 0.00016230198959472835,
185
+ "loss": 0.6978332042694092,
186
+ "mean_token_accuracy": 0.8015618294477462,
187
+ "num_tokens": 423913.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.43452021726010864,
192
+ "eval_entropy": 0.7455583871080634,
193
+ "eval_loss": 0.7118850350379944,
194
+ "eval_mean_token_accuracy": 0.7999682195400923,
195
+ "eval_num_tokens": 423913.0,
196
+ "eval_runtime": 86.7842,
197
+ "eval_samples_per_second": 16.362,
198
+ "eval_steps_per_second": 2.051,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.7810172818601131,
203
+ "epoch": 0.4828002414001207,
204
+ "grad_norm": 0.9276663064956665,
205
+ "learning_rate": 0.00018043629010810582,
206
+ "loss": 0.6937861442565918,
207
+ "mean_token_accuracy": 0.802893053740263,
208
+ "num_tokens": 472395.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.4828002414001207,
213
+ "eval_entropy": 0.7650193290764027,
214
+ "eval_loss": 0.7020567059516907,
215
+ "eval_mean_token_accuracy": 0.8045302153973097,
216
+ "eval_num_tokens": 472395.0,
217
+ "eval_runtime": 88.326,
218
+ "eval_samples_per_second": 16.077,
219
+ "eval_steps_per_second": 2.015,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.7862060204148292,
224
+ "epoch": 0.5310802655401328,
225
+ "grad_norm": 0.8195295929908752,
226
+ "learning_rate": 0.0001985705906214833,
227
+ "loss": 0.6878782272338867,
228
+ "mean_token_accuracy": 0.8027168907225132,
229
+ "num_tokens": 517049.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.5310802655401328,
234
+ "eval_entropy": 0.7656565583154057,
235
+ "eval_loss": 0.6973932981491089,
236
+ "eval_mean_token_accuracy": 0.8047967278555538,
237
+ "eval_num_tokens": 517049.0,
238
+ "eval_runtime": 87.7698,
239
+ "eval_samples_per_second": 16.179,
240
+ "eval_steps_per_second": 2.028,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.7611534461379051,
245
+ "epoch": 0.5793602896801449,
246
+ "grad_norm": 0.7499712109565735,
247
+ "learning_rate": 0.00021670489113486077,
248
+ "loss": 0.681937837600708,
249
+ "mean_token_accuracy": 0.8041978515684605,
250
+ "num_tokens": 562836.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.5793602896801449,
255
+ "eval_entropy": 0.7599442380197933,
256
+ "eval_loss": 0.7002046704292297,
257
+ "eval_mean_token_accuracy": 0.8051861517884759,
258
+ "eval_num_tokens": 562836.0,
259
+ "eval_runtime": 87.9239,
260
+ "eval_samples_per_second": 16.15,
261
+ "eval_steps_per_second": 2.024,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.7800783462822437,
266
+ "epoch": 0.627640313820157,
267
+ "grad_norm": 0.8645269274711609,
268
+ "learning_rate": 0.00023483919164823825,
269
+ "loss": 0.6954500675201416,
270
+ "mean_token_accuracy": 0.8031484372913837,
271
+ "num_tokens": 606612.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.627640313820157,
276
+ "eval_entropy": 0.7498679090751691,
277
+ "eval_loss": 0.7046149373054504,
278
+ "eval_mean_token_accuracy": 0.802388215667746,
279
+ "eval_num_tokens": 606612.0,
280
+ "eval_runtime": 88.2468,
281
+ "eval_samples_per_second": 16.091,
282
+ "eval_steps_per_second": 2.017,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.7586533665657044,
287
+ "epoch": 0.6759203379601689,
288
+ "grad_norm": 0.741596519947052,
289
+ "learning_rate": 0.0002529734921616157,
290
+ "loss": 0.690476369857788,
291
+ "mean_token_accuracy": 0.8046923100948333,
292
+ "num_tokens": 655920.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6759203379601689,
297
+ "eval_entropy": 0.7714274066887544,
298
+ "eval_loss": 0.6992160677909851,
299
+ "eval_mean_token_accuracy": 0.8042646482419432,
300
+ "eval_num_tokens": 655920.0,
301
+ "eval_runtime": 88.1754,
302
+ "eval_samples_per_second": 16.104,
303
+ "eval_steps_per_second": 2.019,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.7585848286747933,
308
+ "epoch": 0.724200362100181,
309
+ "grad_norm": 1.0778124332427979,
310
+ "learning_rate": 0.00027110779267499317,
311
+ "loss": 0.6790409564971924,
312
+ "mean_token_accuracy": 0.8060351841151714,
313
+ "num_tokens": 702638.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.724200362100181,
318
+ "eval_entropy": 0.7380402811457601,
319
+ "eval_loss": 0.7054820656776428,
320
+ "eval_mean_token_accuracy": 0.8003534738267406,
321
+ "eval_num_tokens": 702638.0,
322
+ "eval_runtime": 88.238,
323
+ "eval_samples_per_second": 16.093,
324
+ "eval_steps_per_second": 2.017,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.7696981988847256,
329
+ "epoch": 0.7724803862401931,
330
+ "grad_norm": 0.8408658504486084,
331
+ "learning_rate": 0.00028924209318837064,
332
+ "loss": 0.7070387363433838,
333
+ "mean_token_accuracy": 0.8035947173833847,
334
+ "num_tokens": 749377.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.7724803862401931,
339
+ "eval_entropy": 0.7897409023193831,
340
+ "eval_loss": 0.7015668749809265,
341
+ "eval_mean_token_accuracy": 0.805306687113944,
342
+ "eval_num_tokens": 749377.0,
343
+ "eval_runtime": 87.9395,
344
+ "eval_samples_per_second": 16.147,
345
+ "eval_steps_per_second": 2.024,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.762337189912796,
350
+ "epoch": 0.8207604103802052,
351
+ "grad_norm": 0.9843310117721558,
352
+ "learning_rate": 0.0003073763937017481,
353
+ "loss": 0.7021088123321533,
354
+ "mean_token_accuracy": 0.8029616877436638,
355
+ "num_tokens": 798874.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8207604103802052,
360
+ "eval_entropy": 0.7210077121016685,
361
+ "eval_loss": 0.7082746624946594,
362
+ "eval_mean_token_accuracy": 0.8049283824609906,
363
+ "eval_num_tokens": 798874.0,
364
+ "eval_runtime": 88.1706,
365
+ "eval_samples_per_second": 16.105,
366
+ "eval_steps_per_second": 2.019,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.7879025422036647,
371
+ "epoch": 0.8690404345202173,
372
+ "grad_norm": 0.9713814854621887,
373
+ "learning_rate": 0.0003255106942151256,
374
+ "loss": 0.714624547958374,
375
+ "mean_token_accuracy": 0.79987031519413,
376
+ "num_tokens": 840825.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8690404345202173,
381
+ "eval_entropy": 0.7866930489459735,
382
+ "eval_loss": 0.7145519852638245,
383
+ "eval_mean_token_accuracy": 0.8015657988157165,
384
+ "eval_num_tokens": 840825.0,
385
+ "eval_runtime": 88.3746,
386
+ "eval_samples_per_second": 16.068,
387
+ "eval_steps_per_second": 2.014,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.7844726584851742,
392
+ "epoch": 0.9173204586602294,
393
+ "grad_norm": 1.2028311491012573,
394
+ "learning_rate": 0.00034364499472850306,
395
+ "loss": 0.7040786266326904,
396
+ "mean_token_accuracy": 0.7981885217130185,
397
+ "num_tokens": 882885.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9173204586602294,
402
+ "eval_entropy": 0.7767668339643585,
403
+ "eval_loss": 0.7073574066162109,
404
+ "eval_mean_token_accuracy": 0.8040851833445303,
405
+ "eval_num_tokens": 882885.0,
406
+ "eval_runtime": 88.3828,
407
+ "eval_samples_per_second": 16.066,
408
+ "eval_steps_per_second": 2.014,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.7699790760874748,
413
+ "epoch": 0.9656004828002414,
414
+ "grad_norm": 0.9397398233413696,
415
+ "learning_rate": 0.00036177929524188054,
416
+ "loss": 0.7040313720703125,
417
+ "mean_token_accuracy": 0.8034727744758129,
418
+ "num_tokens": 927456.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9656004828002414,
423
+ "eval_entropy": 0.7749103545472863,
424
+ "eval_loss": 0.7093513011932373,
425
+ "eval_mean_token_accuracy": 0.8019482901926791,
426
+ "eval_num_tokens": 927456.0,
427
+ "eval_runtime": 86.2112,
428
+ "eval_samples_per_second": 16.471,
429
+ "eval_steps_per_second": 2.065,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.7672389172888422,
434
+ "epoch": 1.012070006035003,
435
+ "grad_norm": 0.99210524559021,
436
+ "learning_rate": 0.00037628567078152296,
437
+ "loss": 0.6913118362426758,
438
+ "mean_token_accuracy": 0.8042210944287189,
439
+ "num_tokens": 972662.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.012070006035003,
444
+ "eval_entropy": 0.7294771229283193,
445
+ "eval_loss": 0.7171520590782166,
446
+ "eval_mean_token_accuracy": 0.8015128585059991,
447
+ "eval_num_tokens": 972662.0,
448
+ "eval_runtime": 88.037,
449
+ "eval_samples_per_second": 16.13,
450
+ "eval_steps_per_second": 2.022,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.7143943183124065,
455
+ "epoch": 1.060350030175015,
456
+ "grad_norm": 1.087638258934021,
457
+ "learning_rate": 0.0003762484015601069,
458
+ "loss": 0.642704439163208,
459
+ "mean_token_accuracy": 0.8124966286122799,
460
+ "num_tokens": 1022525.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.060350030175015,
465
+ "eval_entropy": 0.743511886744017,
466
+ "eval_loss": 0.7116662859916687,
467
+ "eval_mean_token_accuracy": 0.8037210672758939,
468
+ "eval_num_tokens": 1022525.0,
469
+ "eval_runtime": 87.9931,
470
+ "eval_samples_per_second": 16.138,
471
+ "eval_steps_per_second": 2.023,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.7237346574664116,
476
+ "epoch": 1.1086300543150271,
477
+ "grad_norm": 0.9365553259849548,
478
+ "learning_rate": 0.0003761579008392123,
479
+ "loss": 0.652644681930542,
480
+ "mean_token_accuracy": 0.8107496216893196,
481
+ "num_tokens": 1069194.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1086300543150271,
486
+ "eval_entropy": 0.7372823839776972,
487
+ "eval_loss": 0.7238303422927856,
488
+ "eval_mean_token_accuracy": 0.8026230124275336,
489
+ "eval_num_tokens": 1069194.0,
490
+ "eval_runtime": 87.6778,
491
+ "eval_samples_per_second": 16.196,
492
+ "eval_steps_per_second": 2.03,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.7202137842774391,
497
+ "epoch": 1.1569100784550392,
498
+ "grad_norm": 0.7998443841934204,
499
+ "learning_rate": 0.0003760141942294726,
500
+ "loss": 0.6485219478607178,
501
+ "mean_token_accuracy": 0.8092714451253414,
502
+ "num_tokens": 1117570.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1569100784550392,
507
+ "eval_entropy": 0.71547647942318,
508
+ "eval_loss": 0.7124277949333191,
509
+ "eval_mean_token_accuracy": 0.8013174219077892,
510
+ "eval_num_tokens": 1117570.0,
511
+ "eval_runtime": 87.4387,
512
+ "eval_samples_per_second": 16.24,
513
+ "eval_steps_per_second": 2.036,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.7270819500088692,
518
+ "epoch": 1.2051901025950513,
519
+ "grad_norm": 0.9470787644386292,
520
+ "learning_rate": 0.00037581732239815854,
521
+ "loss": 0.66352219581604,
522
+ "mean_token_accuracy": 0.8102593503892421,
523
+ "num_tokens": 1161717.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.2051901025950513,
528
+ "eval_entropy": 0.7791565897759427,
529
+ "eval_loss": 0.7179726958274841,
530
+ "eval_mean_token_accuracy": 0.7948898328154275,
531
+ "eval_num_tokens": 1161717.0,
532
+ "eval_runtime": 87.9927,
533
+ "eval_samples_per_second": 16.138,
534
+ "eval_steps_per_second": 2.023,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.7320936664938926,
539
+ "epoch": 1.2534701267350634,
540
+ "grad_norm": 1.0220379829406738,
541
+ "learning_rate": 0.0003755673410576695,
542
+ "loss": 0.6581549644470215,
543
+ "mean_token_accuracy": 0.8098709337413311,
544
+ "num_tokens": 1209390.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.2534701267350634,
549
+ "eval_entropy": 0.7710678225822662,
550
+ "eval_loss": 0.708797812461853,
551
+ "eval_mean_token_accuracy": 0.8039572778042783,
552
+ "eval_num_tokens": 1209390.0,
553
+ "eval_runtime": 88.1193,
554
+ "eval_samples_per_second": 16.115,
555
+ "eval_steps_per_second": 2.02,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.7281674664467573,
560
+ "epoch": 1.3017501508750755,
561
+ "grad_norm": 0.8775188326835632,
562
+ "learning_rate": 0.000375264320949768,
563
+ "loss": 0.6631278991699219,
564
+ "mean_token_accuracy": 0.8101261422038079,
565
+ "num_tokens": 1256611.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3017501508750755,
570
+ "eval_entropy": 0.7199799800187014,
571
+ "eval_loss": 0.6981531381607056,
572
+ "eval_mean_token_accuracy": 0.8071265254127845,
573
+ "eval_num_tokens": 1256611.0,
574
+ "eval_runtime": 88.165,
575
+ "eval_samples_per_second": 16.106,
576
+ "eval_steps_per_second": 2.019,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.722860773652792,
581
+ "epoch": 1.3500301750150876,
582
+ "grad_norm": 0.8949288129806519,
583
+ "learning_rate": 0.00037490834782556,
584
+ "loss": 0.6639598846435547,
585
+ "mean_token_accuracy": 0.8105649061501026,
586
+ "num_tokens": 1302325.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3500301750150876,
591
+ "eval_entropy": 0.7134760032209118,
592
+ "eval_loss": 0.7029948234558105,
593
+ "eval_mean_token_accuracy": 0.8062985880991046,
594
+ "eval_num_tokens": 1302325.0,
595
+ "eval_runtime": 88.1286,
596
+ "eval_samples_per_second": 16.113,
597
+ "eval_steps_per_second": 2.02,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.7299126699566841,
602
+ "epoch": 1.3983101991550995,
603
+ "grad_norm": 1.0323039293289185,
604
+ "learning_rate": 0.0003744995224212291,
605
+ "loss": 0.6535763263702392,
606
+ "mean_token_accuracy": 0.8120224848389626,
607
+ "num_tokens": 1343844.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.3983101991550995,
612
+ "eval_entropy": 0.7169701182440426,
613
+ "eval_loss": 0.7067857980728149,
614
+ "eval_mean_token_accuracy": 0.8040432936689826,
615
+ "eval_num_tokens": 1343844.0,
616
+ "eval_runtime": 88.2252,
617
+ "eval_samples_per_second": 16.095,
618
+ "eval_steps_per_second": 2.018,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.7281742256134749,
623
+ "epoch": 1.4465902232951118,
624
+ "grad_norm": 1.3370424509048462,
625
+ "learning_rate": 0.00037403796042952863,
626
+ "loss": 0.6637272357940673,
627
+ "mean_token_accuracy": 0.8118261776864528,
628
+ "num_tokens": 1389471.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4465902232951118,
633
+ "eval_entropy": 0.7378843212395572,
634
+ "eval_loss": 0.7038390040397644,
635
+ "eval_mean_token_accuracy": 0.8059229371922739,
636
+ "eval_num_tokens": 1389471.0,
637
+ "eval_runtime": 87.9301,
638
+ "eval_samples_per_second": 16.149,
639
+ "eval_steps_per_second": 2.024,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.7388614989817143,
644
+ "epoch": 1.4948702474351236,
645
+ "grad_norm": 0.8464171290397644,
646
+ "learning_rate": 0.00037352379246704257,
647
+ "loss": 0.6604740142822265,
648
+ "mean_token_accuracy": 0.8102974124252796,
649
+ "num_tokens": 1435975.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.4948702474351236,
654
+ "eval_entropy": 0.7126847046814607,
655
+ "eval_loss": 0.6950703263282776,
656
+ "eval_mean_token_accuracy": 0.8085885392815879,
657
+ "eval_num_tokens": 1435975.0,
658
+ "eval_runtime": 88.2732,
659
+ "eval_samples_per_second": 16.086,
660
+ "eval_steps_per_second": 2.016,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.7258916139602661,
665
+ "epoch": 1.5431502715751357,
666
+ "grad_norm": 1.1780085563659668,
667
+ "learning_rate": 0.0003729571640372223,
668
+ "loss": 0.6529891014099121,
669
+ "mean_token_accuracy": 0.8085981778800487,
670
+ "num_tokens": 1482035.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.5431502715751357,
675
+ "eval_entropy": 0.7096509638797032,
676
+ "eval_loss": 0.690051257610321,
677
+ "eval_mean_token_accuracy": 0.8100520241796301,
678
+ "eval_num_tokens": 1482035.0,
679
+ "eval_runtime": 88.3153,
680
+ "eval_samples_per_second": 16.079,
681
+ "eval_steps_per_second": 2.016,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.7151173129677773,
686
+ "epoch": 1.5914302957151478,
687
+ "grad_norm": 1.0357667207717896,
688
+ "learning_rate": 0.0003723382354892108,
689
+ "loss": 0.6451474189758301,
690
+ "mean_token_accuracy": 0.8136902332305909,
691
+ "num_tokens": 1528119.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.5914302957151478,
696
+ "eval_entropy": 0.7060252652409371,
697
+ "eval_loss": 0.6899483799934387,
698
+ "eval_mean_token_accuracy": 0.810834194502134,
699
+ "eval_num_tokens": 1528119.0,
700
+ "eval_runtime": 87.8848,
701
+ "eval_samples_per_second": 16.158,
702
+ "eval_steps_per_second": 2.025,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.7141377851366997,
707
+ "epoch": 1.63971031985516,
708
+ "grad_norm": 0.8391667008399963,
709
+ "learning_rate": 0.000371667181972466,
710
+ "loss": 0.6481579780578614,
711
+ "mean_token_accuracy": 0.8159015104174614,
712
+ "num_tokens": 1573661.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.63971031985516,
717
+ "eval_entropy": 0.7598993319473909,
718
+ "eval_loss": 0.681204080581665,
719
+ "eval_mean_token_accuracy": 0.8100194586126992,
720
+ "eval_num_tokens": 1573661.0,
721
+ "eval_runtime": 88.2415,
722
+ "eval_samples_per_second": 16.092,
723
+ "eval_steps_per_second": 2.017,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.7235980287194252,
728
+ "epoch": 1.687990343995172,
729
+ "grad_norm": 0.8197196125984192,
730
+ "learning_rate": 0.00037094419338719537,
731
+ "loss": 0.6582849025726318,
732
+ "mean_token_accuracy": 0.8114834539592266,
733
+ "num_tokens": 1621714.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.687990343995172,
738
+ "eval_entropy": 0.7166854568411795,
739
+ "eval_loss": 0.679177463054657,
740
+ "eval_mean_token_accuracy": 0.8119039877077167,
741
+ "eval_num_tokens": 1621714.0,
742
+ "eval_runtime": 88.0957,
743
+ "eval_samples_per_second": 16.119,
744
+ "eval_steps_per_second": 2.021,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.7278190158307553,
749
+ "epoch": 1.736270368135184,
750
+ "grad_norm": 0.7820602655410767,
751
+ "learning_rate": 0.0003701694743306164,
752
+ "loss": 0.6633153438568116,
753
+ "mean_token_accuracy": 0.8123249113559723,
754
+ "num_tokens": 1672118.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.736270368135184,
759
+ "eval_entropy": 0.6980850314826108,
760
+ "eval_loss": 0.6775653958320618,
761
+ "eval_mean_token_accuracy": 0.8116872906684875,
762
+ "eval_num_tokens": 1672118.0,
763
+ "eval_runtime": 87.9795,
764
+ "eval_samples_per_second": 16.14,
765
+ "eval_steps_per_second": 2.023,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.7149631556123495,
770
+ "epoch": 1.7845503922751962,
771
+ "grad_norm": 1.0139760971069336,
772
+ "learning_rate": 0.00036934324403905795,
773
+ "loss": 0.6468976020812989,
774
+ "mean_token_accuracy": 0.8117561548948288,
775
+ "num_tokens": 1720465.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.7845503922751962,
780
+ "eval_entropy": 0.69009849094273,
781
+ "eval_loss": 0.6731093525886536,
782
+ "eval_mean_token_accuracy": 0.81281964196248,
783
+ "eval_num_tokens": 1720465.0,
784
+ "eval_runtime": 87.7412,
785
+ "eval_samples_per_second": 16.184,
786
+ "eval_steps_per_second": 2.029,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.7226049453020096,
791
+ "epoch": 1.832830416415208,
792
+ "grad_norm": 1.5040546655654907,
793
+ "learning_rate": 0.0003684657363259193,
794
+ "loss": 0.6385328769683838,
795
+ "mean_token_accuracy": 0.8132639616727829,
796
+ "num_tokens": 1764440.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.832830416415208,
801
+ "eval_entropy": 0.7125169618076153,
802
+ "eval_loss": 0.6864680647850037,
803
+ "eval_mean_token_accuracy": 0.8049418089095126,
804
+ "eval_num_tokens": 1764440.0,
805
+ "eval_runtime": 87.9429,
806
+ "eval_samples_per_second": 16.147,
807
+ "eval_steps_per_second": 2.024,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.707821810990572,
812
+ "epoch": 1.8811104405552204,
813
+ "grad_norm": 0.8604223132133484,
814
+ "learning_rate": 0.00036753719951550327,
815
+ "loss": 0.6397049427032471,
816
+ "mean_token_accuracy": 0.8137877315282822,
817
+ "num_tokens": 1810828.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.8811104405552204,
822
+ "eval_entropy": 0.6492102551326323,
823
+ "eval_loss": 0.6743726134300232,
824
+ "eval_mean_token_accuracy": 0.8130356572317273,
825
+ "eval_num_tokens": 1810828.0,
826
+ "eval_runtime": 87.8684,
827
+ "eval_samples_per_second": 16.161,
828
+ "eval_steps_per_second": 2.026,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.6800355084240437,
833
+ "epoch": 1.9293904646952322,
834
+ "grad_norm": 0.7270982265472412,
835
+ "learning_rate": 0.00036655789637274377,
836
+ "loss": 0.6267755031585693,
837
+ "mean_token_accuracy": 0.8180945307016373,
838
+ "num_tokens": 1857600.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9293904646952322,
843
+ "eval_entropy": 0.6727538651294922,
844
+ "eval_loss": 0.6696028709411621,
845
+ "eval_mean_token_accuracy": 0.8116443090224534,
846
+ "eval_num_tokens": 1857600.0,
847
+ "eval_runtime": 88.2449,
848
+ "eval_samples_per_second": 16.092,
849
+ "eval_steps_per_second": 2.017,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.6746691606938839,
854
+ "epoch": 1.9776704888352445,
855
+ "grad_norm": 0.9771540760993958,
856
+ "learning_rate": 0.0003655281040288461,
857
+ "loss": 0.6326992988586426,
858
+ "mean_token_accuracy": 0.8179976396262646,
859
+ "num_tokens": 1901998.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 1.9776704888352445,
864
+ "eval_entropy": 0.6636319716324967,
865
+ "eval_loss": 0.6714363098144531,
866
+ "eval_mean_token_accuracy": 0.8126635943235976,
867
+ "eval_num_tokens": 1901998.0,
868
+ "eval_runtime": 87.7666,
869
+ "eval_samples_per_second": 16.179,
870
+ "eval_steps_per_second": 2.028,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.6319125778489298,
875
+ "epoch": 2.024140012070006,
876
+ "grad_norm": 0.8380444645881653,
877
+ "learning_rate": 0.0003644481139028622,
878
+ "loss": 0.5780903816223144,
879
+ "mean_token_accuracy": 0.8250188401767186,
880
+ "num_tokens": 1947553.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.024140012070006,
885
+ "eval_entropy": 0.6627151989535,
886
+ "eval_loss": 0.6774935722351074,
887
+ "eval_mean_token_accuracy": 0.8140510819601209,
888
+ "eval_num_tokens": 1947553.0,
889
+ "eval_runtime": 87.8287,
890
+ "eval_samples_per_second": 16.168,
891
+ "eval_steps_per_second": 2.027,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.6059147048741579,
896
+ "epoch": 2.0724200362100182,
897
+ "grad_norm": 0.9745152592658997,
898
+ "learning_rate": 0.0003633182316192229,
899
+ "loss": 0.546702527999878,
900
+ "mean_token_accuracy": 0.8358186542987823,
901
+ "num_tokens": 1995234.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.0724200362100182,
906
+ "eval_entropy": 0.6620710384979677,
907
+ "eval_loss": 0.6769291162490845,
908
+ "eval_mean_token_accuracy": 0.8135082835561773,
909
+ "eval_num_tokens": 1995234.0,
910
+ "eval_runtime": 88.0008,
911
+ "eval_samples_per_second": 16.136,
912
+ "eval_steps_per_second": 2.023,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.6076117843389511,
917
+ "epoch": 2.12070006035003,
918
+ "grad_norm": 0.6914380192756653,
919
+ "learning_rate": 0.0003621387769212491,
920
+ "loss": 0.5324535846710206,
921
+ "mean_token_accuracy": 0.8366568259894848,
922
+ "num_tokens": 2040639.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.12070006035003,
927
+ "eval_entropy": 0.6592224213514435,
928
+ "eval_loss": 0.670383632183075,
929
+ "eval_mean_token_accuracy": 0.8162150088320957,
930
+ "eval_num_tokens": 2040639.0,
931
+ "eval_runtime": 88.324,
932
+ "eval_samples_per_second": 16.077,
933
+ "eval_steps_per_second": 2.015,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.6158072650432587,
938
+ "epoch": 2.1689800844900424,
939
+ "grad_norm": 0.9649198055267334,
940
+ "learning_rate": 0.0003609100835806688,
941
+ "loss": 0.5564568996429443,
942
+ "mean_token_accuracy": 0.8329654954373836,
943
+ "num_tokens": 2090604.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.1689800844900424,
948
+ "eval_entropy": 0.608588819758276,
949
+ "eval_loss": 0.6745102405548096,
950
+ "eval_mean_token_accuracy": 0.8155080542135774,
951
+ "eval_num_tokens": 2090604.0,
952
+ "eval_runtime": 87.52,
953
+ "eval_samples_per_second": 16.225,
954
+ "eval_steps_per_second": 2.034,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.5938734326511621,
959
+ "epoch": 2.2172601086300543,
960
+ "grad_norm": 0.6915557384490967,
961
+ "learning_rate": 0.0003596324993031632,
962
+ "loss": 0.5351875782012939,
963
+ "mean_token_accuracy": 0.8385631121695042,
964
+ "num_tokens": 2142246.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2172601086300543,
969
+ "eval_entropy": 0.6251631205001574,
970
+ "eval_loss": 0.6664518117904663,
971
+ "eval_mean_token_accuracy": 0.81711940417129,
972
+ "eval_num_tokens": 2142246.0,
973
+ "eval_runtime": 87.1792,
974
+ "eval_samples_per_second": 16.288,
975
+ "eval_steps_per_second": 2.042,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.623173613846302,
980
+ "epoch": 2.2655401327700666,
981
+ "grad_norm": 0.8366610407829285,
982
+ "learning_rate": 0.00035830638562997063,
983
+ "loss": 0.5488213539123535,
984
+ "mean_token_accuracy": 0.8321483485400677,
985
+ "num_tokens": 2183144.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.2655401327700666,
990
+ "eval_entropy": 0.6237085649136747,
991
+ "eval_loss": 0.68075031042099,
992
+ "eval_mean_token_accuracy": 0.814671738429016,
993
+ "eval_num_tokens": 2183144.0,
994
+ "eval_runtime": 88.0652,
995
+ "eval_samples_per_second": 16.124,
996
+ "eval_steps_per_second": 2.021,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.6236911740154027,
1001
+ "epoch": 2.3138201569100785,
1002
+ "grad_norm": 0.7812452912330627,
1003
+ "learning_rate": 0.00035693211783557416,
1004
+ "loss": 0.5696200847625732,
1005
+ "mean_token_accuracy": 0.82991396561265,
1006
+ "num_tokens": 2228098.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3138201569100785,
1011
+ "eval_entropy": 0.6327033426319615,
1012
+ "eval_loss": 0.6742984652519226,
1013
+ "eval_mean_token_accuracy": 0.8148086091105857,
1014
+ "eval_num_tokens": 2228098.0,
1015
+ "eval_runtime": 88.1711,
1016
+ "eval_samples_per_second": 16.105,
1017
+ "eval_steps_per_second": 2.019,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.6218426413834095,
1022
+ "epoch": 2.3621001810500903,
1023
+ "grad_norm": 0.8719790577888489,
1024
+ "learning_rate": 0.0003555100848215035,
1025
+ "loss": 0.5582651615142822,
1026
+ "mean_token_accuracy": 0.8323961742222309,
1027
+ "num_tokens": 2273062.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.3621001810500903,
1032
+ "eval_entropy": 0.6338012375858393,
1033
+ "eval_loss": 0.6685822606086731,
1034
+ "eval_mean_token_accuracy": 0.8152180300669724,
1035
+ "eval_num_tokens": 2273062.0,
1036
+ "eval_runtime": 88.101,
1037
+ "eval_samples_per_second": 16.118,
1038
+ "eval_steps_per_second": 2.02,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.6050681818276644,
1043
+ "epoch": 2.4103802051901027,
1044
+ "grad_norm": 0.9151566028594971,
1045
+ "learning_rate": 0.00035404068900628076,
1046
+ "loss": 0.559452486038208,
1047
+ "mean_token_accuracy": 0.8310242861509323,
1048
+ "num_tokens": 2319305.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.4103802051901027,
1053
+ "eval_entropy": 0.6634890969549672,
1054
+ "eval_loss": 0.6692465543746948,
1055
+ "eval_mean_token_accuracy": 0.8134353673190213,
1056
+ "eval_num_tokens": 2319305.0,
1057
+ "eval_runtime": 87.7711,
1058
+ "eval_samples_per_second": 16.178,
1059
+ "eval_steps_per_second": 2.028,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.6169968567788601,
1064
+ "epoch": 2.4586602293301145,
1065
+ "grad_norm": 0.7081006765365601,
1066
+ "learning_rate": 0.0003525243462115402,
1067
+ "loss": 0.5589694023132324,
1068
+ "mean_token_accuracy": 0.8357502184808254,
1069
+ "num_tokens": 2368592.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.4586602293301145,
1074
+ "eval_entropy": 0.6126393578695447,
1075
+ "eval_loss": 0.6706712245941162,
1076
+ "eval_mean_token_accuracy": 0.8155577500884452,
1077
+ "eval_num_tokens": 2368592.0,
1078
+ "eval_runtime": 88.1407,
1079
+ "eval_samples_per_second": 16.111,
1080
+ "eval_steps_per_second": 2.019,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.6072248566895724,
1085
+ "epoch": 2.506940253470127,
1086
+ "grad_norm": 0.6999370455741882,
1087
+ "learning_rate": 0.0003509614855443561,
1088
+ "loss": 0.5500593185424805,
1089
+ "mean_token_accuracy": 0.8337264291942119,
1090
+ "num_tokens": 2412859.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.506940253470127,
1095
+ "eval_entropy": 0.6417417074187418,
1096
+ "eval_loss": 0.6651163697242737,
1097
+ "eval_mean_token_accuracy": 0.8159113908081912,
1098
+ "eval_num_tokens": 2412859.0,
1099
+ "eval_runtime": 87.4854,
1100
+ "eval_samples_per_second": 16.231,
1101
+ "eval_steps_per_second": 2.035,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.6301592070609331,
1106
+ "epoch": 2.5552202776101387,
1107
+ "grad_norm": 0.8768910765647888,
1108
+ "learning_rate": 0.00034935254927581064,
1109
+ "loss": 0.5613903999328613,
1110
+ "mean_token_accuracy": 0.8298896946012974,
1111
+ "num_tokens": 2458201.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.5552202776101387,
1116
+ "eval_entropy": 0.6569794751285167,
1117
+ "eval_loss": 0.6668341159820557,
1118
+ "eval_mean_token_accuracy": 0.8126791795987761,
1119
+ "eval_num_tokens": 2458201.0,
1120
+ "eval_runtime": 88.0474,
1121
+ "eval_samples_per_second": 16.128,
1122
+ "eval_steps_per_second": 2.022,
1123
+ "step": 1060
1124
+ }
1125
+ ],
1126
+ "logging_steps": 20,
1127
+ "max_steps": 4150,
1128
+ "num_input_tokens_seen": 0,
1129
+ "num_train_epochs": 10,
1130
+ "save_steps": 20,
1131
+ "stateful_callbacks": {
1132
+ "TrainerControl": {
1133
+ "args": {
1134
+ "should_epoch_stop": false,
1135
+ "should_evaluate": false,
1136
+ "should_log": false,
1137
+ "should_save": true,
1138
+ "should_training_stop": false
1139
+ },
1140
+ "attributes": {}
1141
+ }
1142
+ },
1143
+ "total_flos": 1.0385749388429107e+17,
1144
+ "train_batch_size": 4,
1145
+ "trial_name": null,
1146
+ "trial_params": null
1147
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.03828026524568501,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "v_proj",
34
+ "up_proj",
35
+ "down_proj",
36
+ "o_proj",
37
+ "q_proj",
38
+ "k_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/trainer_state.json ADDED
@@ -0,0 +1,1168 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.603500301750151,
6
+ "eval_steps": 20,
7
+ "global_step": 1080,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 2.172999547421932,
14
+ "epoch": 0.04828002414001207,
15
+ "grad_norm": 2.5071208477020264,
16
+ "learning_rate": 1.7227585487708594e-05,
17
+ "loss": 1.9347333908081055,
18
+ "mean_token_accuracy": 0.6086816020309925,
19
+ "num_tokens": 47778.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.04828002414001207,
24
+ "eval_entropy": 1.6909754791956269,
25
+ "eval_loss": 1.4728385210037231,
26
+ "eval_mean_token_accuracy": 0.6773593797442619,
27
+ "eval_num_tokens": 47778.0,
28
+ "eval_runtime": 90.1265,
29
+ "eval_samples_per_second": 15.756,
30
+ "eval_steps_per_second": 1.975,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.2014626950025558,
35
+ "epoch": 0.09656004828002414,
36
+ "grad_norm": 1.5074543952941895,
37
+ "learning_rate": 3.5361886001086065e-05,
38
+ "loss": 1.0867146492004394,
39
+ "mean_token_accuracy": 0.7285927519202232,
40
+ "num_tokens": 95981.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.09656004828002414,
45
+ "eval_entropy": 1.0304168352250302,
46
+ "eval_loss": 0.9211422204971313,
47
+ "eval_mean_token_accuracy": 0.7613122848312507,
48
+ "eval_num_tokens": 95981.0,
49
+ "eval_runtime": 88.6338,
50
+ "eval_samples_per_second": 16.021,
51
+ "eval_steps_per_second": 2.008,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.9454198732972146,
56
+ "epoch": 0.14484007242003621,
57
+ "grad_norm": 1.7578108310699463,
58
+ "learning_rate": 5.349618651446354e-05,
59
+ "loss": 0.8447297096252442,
60
+ "mean_token_accuracy": 0.7723424732685089,
61
+ "num_tokens": 142784.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.14484007242003621,
66
+ "eval_entropy": 0.8647834474451086,
67
+ "eval_loss": 0.8240737318992615,
68
+ "eval_mean_token_accuracy": 0.7821227014064789,
69
+ "eval_num_tokens": 142784.0,
70
+ "eval_runtime": 88.0135,
71
+ "eval_samples_per_second": 16.134,
72
+ "eval_steps_per_second": 2.022,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.8963550612330436,
77
+ "epoch": 0.19312009656004828,
78
+ "grad_norm": 1.1464931964874268,
79
+ "learning_rate": 7.1630487027841e-05,
80
+ "loss": 0.7979836463928223,
81
+ "mean_token_accuracy": 0.7822489373385906,
82
+ "num_tokens": 187338.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.19312009656004828,
87
+ "eval_entropy": 0.8663296133614657,
88
+ "eval_loss": 0.7752988338470459,
89
+ "eval_mean_token_accuracy": 0.7885957012015782,
90
+ "eval_num_tokens": 187338.0,
91
+ "eval_runtime": 88.284,
92
+ "eval_samples_per_second": 16.084,
93
+ "eval_steps_per_second": 2.016,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.8265103206038475,
98
+ "epoch": 0.24140012070006034,
99
+ "grad_norm": 1.0106662511825562,
100
+ "learning_rate": 8.976478754121848e-05,
101
+ "loss": 0.7297060012817382,
102
+ "mean_token_accuracy": 0.7955138415098191,
103
+ "num_tokens": 236038.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24140012070006034,
108
+ "eval_entropy": 0.8122169536151244,
109
+ "eval_loss": 0.754906177520752,
110
+ "eval_mean_token_accuracy": 0.7917155255092664,
111
+ "eval_num_tokens": 236038.0,
112
+ "eval_runtime": 87.5754,
113
+ "eval_samples_per_second": 16.215,
114
+ "eval_steps_per_second": 2.033,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.8098392844200134,
119
+ "epoch": 0.28968014484007243,
120
+ "grad_norm": 1.2238041162490845,
121
+ "learning_rate": 0.00010789908805459595,
122
+ "loss": 0.7201489448547364,
123
+ "mean_token_accuracy": 0.7975016921758652,
124
+ "num_tokens": 282918.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.28968014484007243,
129
+ "eval_entropy": 0.8418726462326692,
130
+ "eval_loss": 0.7393178939819336,
131
+ "eval_mean_token_accuracy": 0.7936771585700217,
132
+ "eval_num_tokens": 282918.0,
133
+ "eval_runtime": 87.1597,
134
+ "eval_samples_per_second": 16.292,
135
+ "eval_steps_per_second": 2.042,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.8080591082572937,
140
+ "epoch": 0.33796016898008446,
141
+ "grad_norm": 1.011591911315918,
142
+ "learning_rate": 0.00012603338856797343,
143
+ "loss": 0.7147697925567627,
144
+ "mean_token_accuracy": 0.7947175532579422,
145
+ "num_tokens": 331305.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.33796016898008446,
150
+ "eval_entropy": 0.7894941431083037,
151
+ "eval_loss": 0.720219075679779,
152
+ "eval_mean_token_accuracy": 0.801899236239744,
153
+ "eval_num_tokens": 331305.0,
154
+ "eval_runtime": 87.1493,
155
+ "eval_samples_per_second": 16.294,
156
+ "eval_steps_per_second": 2.042,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.7957205109298229,
161
+ "epoch": 0.38624019312009655,
162
+ "grad_norm": 0.8481830954551697,
163
+ "learning_rate": 0.00014416768908135088,
164
+ "loss": 0.7050027847290039,
165
+ "mean_token_accuracy": 0.7995368830859662,
166
+ "num_tokens": 375876.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.38624019312009655,
171
+ "eval_entropy": 0.7746140591883928,
172
+ "eval_loss": 0.7175475358963013,
173
+ "eval_mean_token_accuracy": 0.7994848955213354,
174
+ "eval_num_tokens": 375876.0,
175
+ "eval_runtime": 86.655,
176
+ "eval_samples_per_second": 16.387,
177
+ "eval_steps_per_second": 2.054,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.779565304517746,
182
+ "epoch": 0.43452021726010864,
183
+ "grad_norm": 1.032624363899231,
184
+ "learning_rate": 0.00016230198959472835,
185
+ "loss": 0.6978332042694092,
186
+ "mean_token_accuracy": 0.8015618294477462,
187
+ "num_tokens": 423913.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.43452021726010864,
192
+ "eval_entropy": 0.7455583871080634,
193
+ "eval_loss": 0.7118850350379944,
194
+ "eval_mean_token_accuracy": 0.7999682195400923,
195
+ "eval_num_tokens": 423913.0,
196
+ "eval_runtime": 86.7842,
197
+ "eval_samples_per_second": 16.362,
198
+ "eval_steps_per_second": 2.051,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.7810172818601131,
203
+ "epoch": 0.4828002414001207,
204
+ "grad_norm": 0.9276663064956665,
205
+ "learning_rate": 0.00018043629010810582,
206
+ "loss": 0.6937861442565918,
207
+ "mean_token_accuracy": 0.802893053740263,
208
+ "num_tokens": 472395.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.4828002414001207,
213
+ "eval_entropy": 0.7650193290764027,
214
+ "eval_loss": 0.7020567059516907,
215
+ "eval_mean_token_accuracy": 0.8045302153973097,
216
+ "eval_num_tokens": 472395.0,
217
+ "eval_runtime": 88.326,
218
+ "eval_samples_per_second": 16.077,
219
+ "eval_steps_per_second": 2.015,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.7862060204148292,
224
+ "epoch": 0.5310802655401328,
225
+ "grad_norm": 0.8195295929908752,
226
+ "learning_rate": 0.0001985705906214833,
227
+ "loss": 0.6878782272338867,
228
+ "mean_token_accuracy": 0.8027168907225132,
229
+ "num_tokens": 517049.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.5310802655401328,
234
+ "eval_entropy": 0.7656565583154057,
235
+ "eval_loss": 0.6973932981491089,
236
+ "eval_mean_token_accuracy": 0.8047967278555538,
237
+ "eval_num_tokens": 517049.0,
238
+ "eval_runtime": 87.7698,
239
+ "eval_samples_per_second": 16.179,
240
+ "eval_steps_per_second": 2.028,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.7611534461379051,
245
+ "epoch": 0.5793602896801449,
246
+ "grad_norm": 0.7499712109565735,
247
+ "learning_rate": 0.00021670489113486077,
248
+ "loss": 0.681937837600708,
249
+ "mean_token_accuracy": 0.8041978515684605,
250
+ "num_tokens": 562836.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.5793602896801449,
255
+ "eval_entropy": 0.7599442380197933,
256
+ "eval_loss": 0.7002046704292297,
257
+ "eval_mean_token_accuracy": 0.8051861517884759,
258
+ "eval_num_tokens": 562836.0,
259
+ "eval_runtime": 87.9239,
260
+ "eval_samples_per_second": 16.15,
261
+ "eval_steps_per_second": 2.024,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.7800783462822437,
266
+ "epoch": 0.627640313820157,
267
+ "grad_norm": 0.8645269274711609,
268
+ "learning_rate": 0.00023483919164823825,
269
+ "loss": 0.6954500675201416,
270
+ "mean_token_accuracy": 0.8031484372913837,
271
+ "num_tokens": 606612.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.627640313820157,
276
+ "eval_entropy": 0.7498679090751691,
277
+ "eval_loss": 0.7046149373054504,
278
+ "eval_mean_token_accuracy": 0.802388215667746,
279
+ "eval_num_tokens": 606612.0,
280
+ "eval_runtime": 88.2468,
281
+ "eval_samples_per_second": 16.091,
282
+ "eval_steps_per_second": 2.017,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.7586533665657044,
287
+ "epoch": 0.6759203379601689,
288
+ "grad_norm": 0.741596519947052,
289
+ "learning_rate": 0.0002529734921616157,
290
+ "loss": 0.690476369857788,
291
+ "mean_token_accuracy": 0.8046923100948333,
292
+ "num_tokens": 655920.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6759203379601689,
297
+ "eval_entropy": 0.7714274066887544,
298
+ "eval_loss": 0.6992160677909851,
299
+ "eval_mean_token_accuracy": 0.8042646482419432,
300
+ "eval_num_tokens": 655920.0,
301
+ "eval_runtime": 88.1754,
302
+ "eval_samples_per_second": 16.104,
303
+ "eval_steps_per_second": 2.019,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.7585848286747933,
308
+ "epoch": 0.724200362100181,
309
+ "grad_norm": 1.0778124332427979,
310
+ "learning_rate": 0.00027110779267499317,
311
+ "loss": 0.6790409564971924,
312
+ "mean_token_accuracy": 0.8060351841151714,
313
+ "num_tokens": 702638.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.724200362100181,
318
+ "eval_entropy": 0.7380402811457601,
319
+ "eval_loss": 0.7054820656776428,
320
+ "eval_mean_token_accuracy": 0.8003534738267406,
321
+ "eval_num_tokens": 702638.0,
322
+ "eval_runtime": 88.238,
323
+ "eval_samples_per_second": 16.093,
324
+ "eval_steps_per_second": 2.017,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.7696981988847256,
329
+ "epoch": 0.7724803862401931,
330
+ "grad_norm": 0.8408658504486084,
331
+ "learning_rate": 0.00028924209318837064,
332
+ "loss": 0.7070387363433838,
333
+ "mean_token_accuracy": 0.8035947173833847,
334
+ "num_tokens": 749377.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.7724803862401931,
339
+ "eval_entropy": 0.7897409023193831,
340
+ "eval_loss": 0.7015668749809265,
341
+ "eval_mean_token_accuracy": 0.805306687113944,
342
+ "eval_num_tokens": 749377.0,
343
+ "eval_runtime": 87.9395,
344
+ "eval_samples_per_second": 16.147,
345
+ "eval_steps_per_second": 2.024,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.762337189912796,
350
+ "epoch": 0.8207604103802052,
351
+ "grad_norm": 0.9843310117721558,
352
+ "learning_rate": 0.0003073763937017481,
353
+ "loss": 0.7021088123321533,
354
+ "mean_token_accuracy": 0.8029616877436638,
355
+ "num_tokens": 798874.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8207604103802052,
360
+ "eval_entropy": 0.7210077121016685,
361
+ "eval_loss": 0.7082746624946594,
362
+ "eval_mean_token_accuracy": 0.8049283824609906,
363
+ "eval_num_tokens": 798874.0,
364
+ "eval_runtime": 88.1706,
365
+ "eval_samples_per_second": 16.105,
366
+ "eval_steps_per_second": 2.019,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.7879025422036647,
371
+ "epoch": 0.8690404345202173,
372
+ "grad_norm": 0.9713814854621887,
373
+ "learning_rate": 0.0003255106942151256,
374
+ "loss": 0.714624547958374,
375
+ "mean_token_accuracy": 0.79987031519413,
376
+ "num_tokens": 840825.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8690404345202173,
381
+ "eval_entropy": 0.7866930489459735,
382
+ "eval_loss": 0.7145519852638245,
383
+ "eval_mean_token_accuracy": 0.8015657988157165,
384
+ "eval_num_tokens": 840825.0,
385
+ "eval_runtime": 88.3746,
386
+ "eval_samples_per_second": 16.068,
387
+ "eval_steps_per_second": 2.014,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.7844726584851742,
392
+ "epoch": 0.9173204586602294,
393
+ "grad_norm": 1.2028311491012573,
394
+ "learning_rate": 0.00034364499472850306,
395
+ "loss": 0.7040786266326904,
396
+ "mean_token_accuracy": 0.7981885217130185,
397
+ "num_tokens": 882885.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9173204586602294,
402
+ "eval_entropy": 0.7767668339643585,
403
+ "eval_loss": 0.7073574066162109,
404
+ "eval_mean_token_accuracy": 0.8040851833445303,
405
+ "eval_num_tokens": 882885.0,
406
+ "eval_runtime": 88.3828,
407
+ "eval_samples_per_second": 16.066,
408
+ "eval_steps_per_second": 2.014,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.7699790760874748,
413
+ "epoch": 0.9656004828002414,
414
+ "grad_norm": 0.9397398233413696,
415
+ "learning_rate": 0.00036177929524188054,
416
+ "loss": 0.7040313720703125,
417
+ "mean_token_accuracy": 0.8034727744758129,
418
+ "num_tokens": 927456.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9656004828002414,
423
+ "eval_entropy": 0.7749103545472863,
424
+ "eval_loss": 0.7093513011932373,
425
+ "eval_mean_token_accuracy": 0.8019482901926791,
426
+ "eval_num_tokens": 927456.0,
427
+ "eval_runtime": 86.2112,
428
+ "eval_samples_per_second": 16.471,
429
+ "eval_steps_per_second": 2.065,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.7672389172888422,
434
+ "epoch": 1.012070006035003,
435
+ "grad_norm": 0.99210524559021,
436
+ "learning_rate": 0.00037628567078152296,
437
+ "loss": 0.6913118362426758,
438
+ "mean_token_accuracy": 0.8042210944287189,
439
+ "num_tokens": 972662.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.012070006035003,
444
+ "eval_entropy": 0.7294771229283193,
445
+ "eval_loss": 0.7171520590782166,
446
+ "eval_mean_token_accuracy": 0.8015128585059991,
447
+ "eval_num_tokens": 972662.0,
448
+ "eval_runtime": 88.037,
449
+ "eval_samples_per_second": 16.13,
450
+ "eval_steps_per_second": 2.022,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.7143943183124065,
455
+ "epoch": 1.060350030175015,
456
+ "grad_norm": 1.087638258934021,
457
+ "learning_rate": 0.0003762484015601069,
458
+ "loss": 0.642704439163208,
459
+ "mean_token_accuracy": 0.8124966286122799,
460
+ "num_tokens": 1022525.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.060350030175015,
465
+ "eval_entropy": 0.743511886744017,
466
+ "eval_loss": 0.7116662859916687,
467
+ "eval_mean_token_accuracy": 0.8037210672758939,
468
+ "eval_num_tokens": 1022525.0,
469
+ "eval_runtime": 87.9931,
470
+ "eval_samples_per_second": 16.138,
471
+ "eval_steps_per_second": 2.023,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.7237346574664116,
476
+ "epoch": 1.1086300543150271,
477
+ "grad_norm": 0.9365553259849548,
478
+ "learning_rate": 0.0003761579008392123,
479
+ "loss": 0.652644681930542,
480
+ "mean_token_accuracy": 0.8107496216893196,
481
+ "num_tokens": 1069194.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1086300543150271,
486
+ "eval_entropy": 0.7372823839776972,
487
+ "eval_loss": 0.7238303422927856,
488
+ "eval_mean_token_accuracy": 0.8026230124275336,
489
+ "eval_num_tokens": 1069194.0,
490
+ "eval_runtime": 87.6778,
491
+ "eval_samples_per_second": 16.196,
492
+ "eval_steps_per_second": 2.03,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.7202137842774391,
497
+ "epoch": 1.1569100784550392,
498
+ "grad_norm": 0.7998443841934204,
499
+ "learning_rate": 0.0003760141942294726,
500
+ "loss": 0.6485219478607178,
501
+ "mean_token_accuracy": 0.8092714451253414,
502
+ "num_tokens": 1117570.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1569100784550392,
507
+ "eval_entropy": 0.71547647942318,
508
+ "eval_loss": 0.7124277949333191,
509
+ "eval_mean_token_accuracy": 0.8013174219077892,
510
+ "eval_num_tokens": 1117570.0,
511
+ "eval_runtime": 87.4387,
512
+ "eval_samples_per_second": 16.24,
513
+ "eval_steps_per_second": 2.036,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.7270819500088692,
518
+ "epoch": 1.2051901025950513,
519
+ "grad_norm": 0.9470787644386292,
520
+ "learning_rate": 0.00037581732239815854,
521
+ "loss": 0.66352219581604,
522
+ "mean_token_accuracy": 0.8102593503892421,
523
+ "num_tokens": 1161717.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.2051901025950513,
528
+ "eval_entropy": 0.7791565897759427,
529
+ "eval_loss": 0.7179726958274841,
530
+ "eval_mean_token_accuracy": 0.7948898328154275,
531
+ "eval_num_tokens": 1161717.0,
532
+ "eval_runtime": 87.9927,
533
+ "eval_samples_per_second": 16.138,
534
+ "eval_steps_per_second": 2.023,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.7320936664938926,
539
+ "epoch": 1.2534701267350634,
540
+ "grad_norm": 1.0220379829406738,
541
+ "learning_rate": 0.0003755673410576695,
542
+ "loss": 0.6581549644470215,
543
+ "mean_token_accuracy": 0.8098709337413311,
544
+ "num_tokens": 1209390.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.2534701267350634,
549
+ "eval_entropy": 0.7710678225822662,
550
+ "eval_loss": 0.708797812461853,
551
+ "eval_mean_token_accuracy": 0.8039572778042783,
552
+ "eval_num_tokens": 1209390.0,
553
+ "eval_runtime": 88.1193,
554
+ "eval_samples_per_second": 16.115,
555
+ "eval_steps_per_second": 2.02,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.7281674664467573,
560
+ "epoch": 1.3017501508750755,
561
+ "grad_norm": 0.8775188326835632,
562
+ "learning_rate": 0.000375264320949768,
563
+ "loss": 0.6631278991699219,
564
+ "mean_token_accuracy": 0.8101261422038079,
565
+ "num_tokens": 1256611.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3017501508750755,
570
+ "eval_entropy": 0.7199799800187014,
571
+ "eval_loss": 0.6981531381607056,
572
+ "eval_mean_token_accuracy": 0.8071265254127845,
573
+ "eval_num_tokens": 1256611.0,
574
+ "eval_runtime": 88.165,
575
+ "eval_samples_per_second": 16.106,
576
+ "eval_steps_per_second": 2.019,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.722860773652792,
581
+ "epoch": 1.3500301750150876,
582
+ "grad_norm": 0.8949288129806519,
583
+ "learning_rate": 0.00037490834782556,
584
+ "loss": 0.6639598846435547,
585
+ "mean_token_accuracy": 0.8105649061501026,
586
+ "num_tokens": 1302325.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3500301750150876,
591
+ "eval_entropy": 0.7134760032209118,
592
+ "eval_loss": 0.7029948234558105,
593
+ "eval_mean_token_accuracy": 0.8062985880991046,
594
+ "eval_num_tokens": 1302325.0,
595
+ "eval_runtime": 88.1286,
596
+ "eval_samples_per_second": 16.113,
597
+ "eval_steps_per_second": 2.02,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.7299126699566841,
602
+ "epoch": 1.3983101991550995,
603
+ "grad_norm": 1.0323039293289185,
604
+ "learning_rate": 0.0003744995224212291,
605
+ "loss": 0.6535763263702392,
606
+ "mean_token_accuracy": 0.8120224848389626,
607
+ "num_tokens": 1343844.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.3983101991550995,
612
+ "eval_entropy": 0.7169701182440426,
613
+ "eval_loss": 0.7067857980728149,
614
+ "eval_mean_token_accuracy": 0.8040432936689826,
615
+ "eval_num_tokens": 1343844.0,
616
+ "eval_runtime": 88.2252,
617
+ "eval_samples_per_second": 16.095,
618
+ "eval_steps_per_second": 2.018,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.7281742256134749,
623
+ "epoch": 1.4465902232951118,
624
+ "grad_norm": 1.3370424509048462,
625
+ "learning_rate": 0.00037403796042952863,
626
+ "loss": 0.6637272357940673,
627
+ "mean_token_accuracy": 0.8118261776864528,
628
+ "num_tokens": 1389471.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4465902232951118,
633
+ "eval_entropy": 0.7378843212395572,
634
+ "eval_loss": 0.7038390040397644,
635
+ "eval_mean_token_accuracy": 0.8059229371922739,
636
+ "eval_num_tokens": 1389471.0,
637
+ "eval_runtime": 87.9301,
638
+ "eval_samples_per_second": 16.149,
639
+ "eval_steps_per_second": 2.024,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.7388614989817143,
644
+ "epoch": 1.4948702474351236,
645
+ "grad_norm": 0.8464171290397644,
646
+ "learning_rate": 0.00037352379246704257,
647
+ "loss": 0.6604740142822265,
648
+ "mean_token_accuracy": 0.8102974124252796,
649
+ "num_tokens": 1435975.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.4948702474351236,
654
+ "eval_entropy": 0.7126847046814607,
655
+ "eval_loss": 0.6950703263282776,
656
+ "eval_mean_token_accuracy": 0.8085885392815879,
657
+ "eval_num_tokens": 1435975.0,
658
+ "eval_runtime": 88.2732,
659
+ "eval_samples_per_second": 16.086,
660
+ "eval_steps_per_second": 2.016,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.7258916139602661,
665
+ "epoch": 1.5431502715751357,
666
+ "grad_norm": 1.1780085563659668,
667
+ "learning_rate": 0.0003729571640372223,
668
+ "loss": 0.6529891014099121,
669
+ "mean_token_accuracy": 0.8085981778800487,
670
+ "num_tokens": 1482035.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.5431502715751357,
675
+ "eval_entropy": 0.7096509638797032,
676
+ "eval_loss": 0.690051257610321,
677
+ "eval_mean_token_accuracy": 0.8100520241796301,
678
+ "eval_num_tokens": 1482035.0,
679
+ "eval_runtime": 88.3153,
680
+ "eval_samples_per_second": 16.079,
681
+ "eval_steps_per_second": 2.016,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.7151173129677773,
686
+ "epoch": 1.5914302957151478,
687
+ "grad_norm": 1.0357667207717896,
688
+ "learning_rate": 0.0003723382354892108,
689
+ "loss": 0.6451474189758301,
690
+ "mean_token_accuracy": 0.8136902332305909,
691
+ "num_tokens": 1528119.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.5914302957151478,
696
+ "eval_entropy": 0.7060252652409371,
697
+ "eval_loss": 0.6899483799934387,
698
+ "eval_mean_token_accuracy": 0.810834194502134,
699
+ "eval_num_tokens": 1528119.0,
700
+ "eval_runtime": 87.8848,
701
+ "eval_samples_per_second": 16.158,
702
+ "eval_steps_per_second": 2.025,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.7141377851366997,
707
+ "epoch": 1.63971031985516,
708
+ "grad_norm": 0.8391667008399963,
709
+ "learning_rate": 0.000371667181972466,
710
+ "loss": 0.6481579780578614,
711
+ "mean_token_accuracy": 0.8159015104174614,
712
+ "num_tokens": 1573661.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.63971031985516,
717
+ "eval_entropy": 0.7598993319473909,
718
+ "eval_loss": 0.681204080581665,
719
+ "eval_mean_token_accuracy": 0.8100194586126992,
720
+ "eval_num_tokens": 1573661.0,
721
+ "eval_runtime": 88.2415,
722
+ "eval_samples_per_second": 16.092,
723
+ "eval_steps_per_second": 2.017,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.7235980287194252,
728
+ "epoch": 1.687990343995172,
729
+ "grad_norm": 0.8197196125984192,
730
+ "learning_rate": 0.00037094419338719537,
731
+ "loss": 0.6582849025726318,
732
+ "mean_token_accuracy": 0.8114834539592266,
733
+ "num_tokens": 1621714.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.687990343995172,
738
+ "eval_entropy": 0.7166854568411795,
739
+ "eval_loss": 0.679177463054657,
740
+ "eval_mean_token_accuracy": 0.8119039877077167,
741
+ "eval_num_tokens": 1621714.0,
742
+ "eval_runtime": 88.0957,
743
+ "eval_samples_per_second": 16.119,
744
+ "eval_steps_per_second": 2.021,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.7278190158307553,
749
+ "epoch": 1.736270368135184,
750
+ "grad_norm": 0.7820602655410767,
751
+ "learning_rate": 0.0003701694743306164,
752
+ "loss": 0.6633153438568116,
753
+ "mean_token_accuracy": 0.8123249113559723,
754
+ "num_tokens": 1672118.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.736270368135184,
759
+ "eval_entropy": 0.6980850314826108,
760
+ "eval_loss": 0.6775653958320618,
761
+ "eval_mean_token_accuracy": 0.8116872906684875,
762
+ "eval_num_tokens": 1672118.0,
763
+ "eval_runtime": 87.9795,
764
+ "eval_samples_per_second": 16.14,
765
+ "eval_steps_per_second": 2.023,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.7149631556123495,
770
+ "epoch": 1.7845503922751962,
771
+ "grad_norm": 1.0139760971069336,
772
+ "learning_rate": 0.00036934324403905795,
773
+ "loss": 0.6468976020812989,
774
+ "mean_token_accuracy": 0.8117561548948288,
775
+ "num_tokens": 1720465.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.7845503922751962,
780
+ "eval_entropy": 0.69009849094273,
781
+ "eval_loss": 0.6731093525886536,
782
+ "eval_mean_token_accuracy": 0.81281964196248,
783
+ "eval_num_tokens": 1720465.0,
784
+ "eval_runtime": 87.7412,
785
+ "eval_samples_per_second": 16.184,
786
+ "eval_steps_per_second": 2.029,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.7226049453020096,
791
+ "epoch": 1.832830416415208,
792
+ "grad_norm": 1.5040546655654907,
793
+ "learning_rate": 0.0003684657363259193,
794
+ "loss": 0.6385328769683838,
795
+ "mean_token_accuracy": 0.8132639616727829,
796
+ "num_tokens": 1764440.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.832830416415208,
801
+ "eval_entropy": 0.7125169618076153,
802
+ "eval_loss": 0.6864680647850037,
803
+ "eval_mean_token_accuracy": 0.8049418089095126,
804
+ "eval_num_tokens": 1764440.0,
805
+ "eval_runtime": 87.9429,
806
+ "eval_samples_per_second": 16.147,
807
+ "eval_steps_per_second": 2.024,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.707821810990572,
812
+ "epoch": 1.8811104405552204,
813
+ "grad_norm": 0.8604223132133484,
814
+ "learning_rate": 0.00036753719951550327,
815
+ "loss": 0.6397049427032471,
816
+ "mean_token_accuracy": 0.8137877315282822,
817
+ "num_tokens": 1810828.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.8811104405552204,
822
+ "eval_entropy": 0.6492102551326323,
823
+ "eval_loss": 0.6743726134300232,
824
+ "eval_mean_token_accuracy": 0.8130356572317273,
825
+ "eval_num_tokens": 1810828.0,
826
+ "eval_runtime": 87.8684,
827
+ "eval_samples_per_second": 16.161,
828
+ "eval_steps_per_second": 2.026,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.6800355084240437,
833
+ "epoch": 1.9293904646952322,
834
+ "grad_norm": 0.7270982265472412,
835
+ "learning_rate": 0.00036655789637274377,
836
+ "loss": 0.6267755031585693,
837
+ "mean_token_accuracy": 0.8180945307016373,
838
+ "num_tokens": 1857600.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9293904646952322,
843
+ "eval_entropy": 0.6727538651294922,
844
+ "eval_loss": 0.6696028709411621,
845
+ "eval_mean_token_accuracy": 0.8116443090224534,
846
+ "eval_num_tokens": 1857600.0,
847
+ "eval_runtime": 88.2449,
848
+ "eval_samples_per_second": 16.092,
849
+ "eval_steps_per_second": 2.017,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.6746691606938839,
854
+ "epoch": 1.9776704888352445,
855
+ "grad_norm": 0.9771540760993958,
856
+ "learning_rate": 0.0003655281040288461,
857
+ "loss": 0.6326992988586426,
858
+ "mean_token_accuracy": 0.8179976396262646,
859
+ "num_tokens": 1901998.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 1.9776704888352445,
864
+ "eval_entropy": 0.6636319716324967,
865
+ "eval_loss": 0.6714363098144531,
866
+ "eval_mean_token_accuracy": 0.8126635943235976,
867
+ "eval_num_tokens": 1901998.0,
868
+ "eval_runtime": 87.7666,
869
+ "eval_samples_per_second": 16.179,
870
+ "eval_steps_per_second": 2.028,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.6319125778489298,
875
+ "epoch": 2.024140012070006,
876
+ "grad_norm": 0.8380444645881653,
877
+ "learning_rate": 0.0003644481139028622,
878
+ "loss": 0.5780903816223144,
879
+ "mean_token_accuracy": 0.8250188401767186,
880
+ "num_tokens": 1947553.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.024140012070006,
885
+ "eval_entropy": 0.6627151989535,
886
+ "eval_loss": 0.6774935722351074,
887
+ "eval_mean_token_accuracy": 0.8140510819601209,
888
+ "eval_num_tokens": 1947553.0,
889
+ "eval_runtime": 87.8287,
890
+ "eval_samples_per_second": 16.168,
891
+ "eval_steps_per_second": 2.027,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.6059147048741579,
896
+ "epoch": 2.0724200362100182,
897
+ "grad_norm": 0.9745152592658997,
898
+ "learning_rate": 0.0003633182316192229,
899
+ "loss": 0.546702527999878,
900
+ "mean_token_accuracy": 0.8358186542987823,
901
+ "num_tokens": 1995234.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.0724200362100182,
906
+ "eval_entropy": 0.6620710384979677,
907
+ "eval_loss": 0.6769291162490845,
908
+ "eval_mean_token_accuracy": 0.8135082835561773,
909
+ "eval_num_tokens": 1995234.0,
910
+ "eval_runtime": 88.0008,
911
+ "eval_samples_per_second": 16.136,
912
+ "eval_steps_per_second": 2.023,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.6076117843389511,
917
+ "epoch": 2.12070006035003,
918
+ "grad_norm": 0.6914380192756653,
919
+ "learning_rate": 0.0003621387769212491,
920
+ "loss": 0.5324535846710206,
921
+ "mean_token_accuracy": 0.8366568259894848,
922
+ "num_tokens": 2040639.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.12070006035003,
927
+ "eval_entropy": 0.6592224213514435,
928
+ "eval_loss": 0.670383632183075,
929
+ "eval_mean_token_accuracy": 0.8162150088320957,
930
+ "eval_num_tokens": 2040639.0,
931
+ "eval_runtime": 88.324,
932
+ "eval_samples_per_second": 16.077,
933
+ "eval_steps_per_second": 2.015,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.6158072650432587,
938
+ "epoch": 2.1689800844900424,
939
+ "grad_norm": 0.9649198055267334,
940
+ "learning_rate": 0.0003609100835806688,
941
+ "loss": 0.5564568996429443,
942
+ "mean_token_accuracy": 0.8329654954373836,
943
+ "num_tokens": 2090604.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.1689800844900424,
948
+ "eval_entropy": 0.608588819758276,
949
+ "eval_loss": 0.6745102405548096,
950
+ "eval_mean_token_accuracy": 0.8155080542135774,
951
+ "eval_num_tokens": 2090604.0,
952
+ "eval_runtime": 87.52,
953
+ "eval_samples_per_second": 16.225,
954
+ "eval_steps_per_second": 2.034,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.5938734326511621,
959
+ "epoch": 2.2172601086300543,
960
+ "grad_norm": 0.6915557384490967,
961
+ "learning_rate": 0.0003596324993031632,
962
+ "loss": 0.5351875782012939,
963
+ "mean_token_accuracy": 0.8385631121695042,
964
+ "num_tokens": 2142246.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2172601086300543,
969
+ "eval_entropy": 0.6251631205001574,
970
+ "eval_loss": 0.6664518117904663,
971
+ "eval_mean_token_accuracy": 0.81711940417129,
972
+ "eval_num_tokens": 2142246.0,
973
+ "eval_runtime": 87.1792,
974
+ "eval_samples_per_second": 16.288,
975
+ "eval_steps_per_second": 2.042,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.623173613846302,
980
+ "epoch": 2.2655401327700666,
981
+ "grad_norm": 0.8366610407829285,
982
+ "learning_rate": 0.00035830638562997063,
983
+ "loss": 0.5488213539123535,
984
+ "mean_token_accuracy": 0.8321483485400677,
985
+ "num_tokens": 2183144.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.2655401327700666,
990
+ "eval_entropy": 0.6237085649136747,
991
+ "eval_loss": 0.68075031042099,
992
+ "eval_mean_token_accuracy": 0.814671738429016,
993
+ "eval_num_tokens": 2183144.0,
994
+ "eval_runtime": 88.0652,
995
+ "eval_samples_per_second": 16.124,
996
+ "eval_steps_per_second": 2.021,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.6236911740154027,
1001
+ "epoch": 2.3138201569100785,
1002
+ "grad_norm": 0.7812452912330627,
1003
+ "learning_rate": 0.00035693211783557416,
1004
+ "loss": 0.5696200847625732,
1005
+ "mean_token_accuracy": 0.82991396561265,
1006
+ "num_tokens": 2228098.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3138201569100785,
1011
+ "eval_entropy": 0.6327033426319615,
1012
+ "eval_loss": 0.6742984652519226,
1013
+ "eval_mean_token_accuracy": 0.8148086091105857,
1014
+ "eval_num_tokens": 2228098.0,
1015
+ "eval_runtime": 88.1711,
1016
+ "eval_samples_per_second": 16.105,
1017
+ "eval_steps_per_second": 2.019,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.6218426413834095,
1022
+ "epoch": 2.3621001810500903,
1023
+ "grad_norm": 0.8719790577888489,
1024
+ "learning_rate": 0.0003555100848215035,
1025
+ "loss": 0.5582651615142822,
1026
+ "mean_token_accuracy": 0.8323961742222309,
1027
+ "num_tokens": 2273062.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.3621001810500903,
1032
+ "eval_entropy": 0.6338012375858393,
1033
+ "eval_loss": 0.6685822606086731,
1034
+ "eval_mean_token_accuracy": 0.8152180300669724,
1035
+ "eval_num_tokens": 2273062.0,
1036
+ "eval_runtime": 88.101,
1037
+ "eval_samples_per_second": 16.118,
1038
+ "eval_steps_per_second": 2.02,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.6050681818276644,
1043
+ "epoch": 2.4103802051901027,
1044
+ "grad_norm": 0.9151566028594971,
1045
+ "learning_rate": 0.00035404068900628076,
1046
+ "loss": 0.559452486038208,
1047
+ "mean_token_accuracy": 0.8310242861509323,
1048
+ "num_tokens": 2319305.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.4103802051901027,
1053
+ "eval_entropy": 0.6634890969549672,
1054
+ "eval_loss": 0.6692465543746948,
1055
+ "eval_mean_token_accuracy": 0.8134353673190213,
1056
+ "eval_num_tokens": 2319305.0,
1057
+ "eval_runtime": 87.7711,
1058
+ "eval_samples_per_second": 16.178,
1059
+ "eval_steps_per_second": 2.028,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.6169968567788601,
1064
+ "epoch": 2.4586602293301145,
1065
+ "grad_norm": 0.7081006765365601,
1066
+ "learning_rate": 0.0003525243462115402,
1067
+ "loss": 0.5589694023132324,
1068
+ "mean_token_accuracy": 0.8357502184808254,
1069
+ "num_tokens": 2368592.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.4586602293301145,
1074
+ "eval_entropy": 0.6126393578695447,
1075
+ "eval_loss": 0.6706712245941162,
1076
+ "eval_mean_token_accuracy": 0.8155577500884452,
1077
+ "eval_num_tokens": 2368592.0,
1078
+ "eval_runtime": 88.1407,
1079
+ "eval_samples_per_second": 16.111,
1080
+ "eval_steps_per_second": 2.019,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.6072248566895724,
1085
+ "epoch": 2.506940253470127,
1086
+ "grad_norm": 0.6999370455741882,
1087
+ "learning_rate": 0.0003509614855443561,
1088
+ "loss": 0.5500593185424805,
1089
+ "mean_token_accuracy": 0.8337264291942119,
1090
+ "num_tokens": 2412859.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.506940253470127,
1095
+ "eval_entropy": 0.6417417074187418,
1096
+ "eval_loss": 0.6651163697242737,
1097
+ "eval_mean_token_accuracy": 0.8159113908081912,
1098
+ "eval_num_tokens": 2412859.0,
1099
+ "eval_runtime": 87.4854,
1100
+ "eval_samples_per_second": 16.231,
1101
+ "eval_steps_per_second": 2.035,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.6301592070609331,
1106
+ "epoch": 2.5552202776101387,
1107
+ "grad_norm": 0.8768910765647888,
1108
+ "learning_rate": 0.00034935254927581064,
1109
+ "loss": 0.5613903999328613,
1110
+ "mean_token_accuracy": 0.8298896946012974,
1111
+ "num_tokens": 2458201.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.5552202776101387,
1116
+ "eval_entropy": 0.6569794751285167,
1117
+ "eval_loss": 0.6668341159820557,
1118
+ "eval_mean_token_accuracy": 0.8126791795987761,
1119
+ "eval_num_tokens": 2458201.0,
1120
+ "eval_runtime": 88.0474,
1121
+ "eval_samples_per_second": 16.128,
1122
+ "eval_steps_per_second": 2.022,
1123
+ "step": 1060
1124
+ },
1125
+ {
1126
+ "entropy": 0.6069021724164486,
1127
+ "epoch": 2.603500301750151,
1128
+ "grad_norm": 0.9033508896827698,
1129
+ "learning_rate": 0.0003476979927158357,
1130
+ "loss": 0.5590654373168945,
1131
+ "mean_token_accuracy": 0.831520090252161,
1132
+ "num_tokens": 2505382.0,
1133
+ "step": 1080
1134
+ },
1135
+ {
1136
+ "epoch": 2.603500301750151,
1137
+ "eval_entropy": 0.6636380777600106,
1138
+ "eval_loss": 0.6596238017082214,
1139
+ "eval_mean_token_accuracy": 0.8177202826135614,
1140
+ "eval_num_tokens": 2505382.0,
1141
+ "eval_runtime": 88.08,
1142
+ "eval_samples_per_second": 16.122,
1143
+ "eval_steps_per_second": 2.021,
1144
+ "step": 1080
1145
+ }
1146
+ ],
1147
+ "logging_steps": 20,
1148
+ "max_steps": 4150,
1149
+ "num_input_tokens_seen": 0,
1150
+ "num_train_epochs": 10,
1151
+ "save_steps": 20,
1152
+ "stateful_callbacks": {
1153
+ "TrainerControl": {
1154
+ "args": {
1155
+ "should_epoch_stop": false,
1156
+ "should_evaluate": false,
1157
+ "should_log": false,
1158
+ "should_save": true,
1159
+ "should_training_stop": false
1160
+ },
1161
+ "attributes": {}
1162
+ }
1163
+ },
1164
+ "total_flos": 1.0583776570652467e+17,
1165
+ "train_batch_size": 4,
1166
+ "trial_name": null,
1167
+ "trial_params": null
1168
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.03828026524568501,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "v_proj",
34
+ "up_proj",
35
+ "down_proj",
36
+ "o_proj",
37
+ "q_proj",
38
+ "k_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/trainer_state.json ADDED
@@ -0,0 +1,1189 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.651780325890163,
6
+ "eval_steps": 20,
7
+ "global_step": 1100,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 2.172999547421932,
14
+ "epoch": 0.04828002414001207,
15
+ "grad_norm": 2.5071208477020264,
16
+ "learning_rate": 1.7227585487708594e-05,
17
+ "loss": 1.9347333908081055,
18
+ "mean_token_accuracy": 0.6086816020309925,
19
+ "num_tokens": 47778.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.04828002414001207,
24
+ "eval_entropy": 1.6909754791956269,
25
+ "eval_loss": 1.4728385210037231,
26
+ "eval_mean_token_accuracy": 0.6773593797442619,
27
+ "eval_num_tokens": 47778.0,
28
+ "eval_runtime": 90.1265,
29
+ "eval_samples_per_second": 15.756,
30
+ "eval_steps_per_second": 1.975,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.2014626950025558,
35
+ "epoch": 0.09656004828002414,
36
+ "grad_norm": 1.5074543952941895,
37
+ "learning_rate": 3.5361886001086065e-05,
38
+ "loss": 1.0867146492004394,
39
+ "mean_token_accuracy": 0.7285927519202232,
40
+ "num_tokens": 95981.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.09656004828002414,
45
+ "eval_entropy": 1.0304168352250302,
46
+ "eval_loss": 0.9211422204971313,
47
+ "eval_mean_token_accuracy": 0.7613122848312507,
48
+ "eval_num_tokens": 95981.0,
49
+ "eval_runtime": 88.6338,
50
+ "eval_samples_per_second": 16.021,
51
+ "eval_steps_per_second": 2.008,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.9454198732972146,
56
+ "epoch": 0.14484007242003621,
57
+ "grad_norm": 1.7578108310699463,
58
+ "learning_rate": 5.349618651446354e-05,
59
+ "loss": 0.8447297096252442,
60
+ "mean_token_accuracy": 0.7723424732685089,
61
+ "num_tokens": 142784.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.14484007242003621,
66
+ "eval_entropy": 0.8647834474451086,
67
+ "eval_loss": 0.8240737318992615,
68
+ "eval_mean_token_accuracy": 0.7821227014064789,
69
+ "eval_num_tokens": 142784.0,
70
+ "eval_runtime": 88.0135,
71
+ "eval_samples_per_second": 16.134,
72
+ "eval_steps_per_second": 2.022,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.8963550612330436,
77
+ "epoch": 0.19312009656004828,
78
+ "grad_norm": 1.1464931964874268,
79
+ "learning_rate": 7.1630487027841e-05,
80
+ "loss": 0.7979836463928223,
81
+ "mean_token_accuracy": 0.7822489373385906,
82
+ "num_tokens": 187338.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.19312009656004828,
87
+ "eval_entropy": 0.8663296133614657,
88
+ "eval_loss": 0.7752988338470459,
89
+ "eval_mean_token_accuracy": 0.7885957012015782,
90
+ "eval_num_tokens": 187338.0,
91
+ "eval_runtime": 88.284,
92
+ "eval_samples_per_second": 16.084,
93
+ "eval_steps_per_second": 2.016,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.8265103206038475,
98
+ "epoch": 0.24140012070006034,
99
+ "grad_norm": 1.0106662511825562,
100
+ "learning_rate": 8.976478754121848e-05,
101
+ "loss": 0.7297060012817382,
102
+ "mean_token_accuracy": 0.7955138415098191,
103
+ "num_tokens": 236038.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24140012070006034,
108
+ "eval_entropy": 0.8122169536151244,
109
+ "eval_loss": 0.754906177520752,
110
+ "eval_mean_token_accuracy": 0.7917155255092664,
111
+ "eval_num_tokens": 236038.0,
112
+ "eval_runtime": 87.5754,
113
+ "eval_samples_per_second": 16.215,
114
+ "eval_steps_per_second": 2.033,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.8098392844200134,
119
+ "epoch": 0.28968014484007243,
120
+ "grad_norm": 1.2238041162490845,
121
+ "learning_rate": 0.00010789908805459595,
122
+ "loss": 0.7201489448547364,
123
+ "mean_token_accuracy": 0.7975016921758652,
124
+ "num_tokens": 282918.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.28968014484007243,
129
+ "eval_entropy": 0.8418726462326692,
130
+ "eval_loss": 0.7393178939819336,
131
+ "eval_mean_token_accuracy": 0.7936771585700217,
132
+ "eval_num_tokens": 282918.0,
133
+ "eval_runtime": 87.1597,
134
+ "eval_samples_per_second": 16.292,
135
+ "eval_steps_per_second": 2.042,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.8080591082572937,
140
+ "epoch": 0.33796016898008446,
141
+ "grad_norm": 1.011591911315918,
142
+ "learning_rate": 0.00012603338856797343,
143
+ "loss": 0.7147697925567627,
144
+ "mean_token_accuracy": 0.7947175532579422,
145
+ "num_tokens": 331305.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.33796016898008446,
150
+ "eval_entropy": 0.7894941431083037,
151
+ "eval_loss": 0.720219075679779,
152
+ "eval_mean_token_accuracy": 0.801899236239744,
153
+ "eval_num_tokens": 331305.0,
154
+ "eval_runtime": 87.1493,
155
+ "eval_samples_per_second": 16.294,
156
+ "eval_steps_per_second": 2.042,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.7957205109298229,
161
+ "epoch": 0.38624019312009655,
162
+ "grad_norm": 0.8481830954551697,
163
+ "learning_rate": 0.00014416768908135088,
164
+ "loss": 0.7050027847290039,
165
+ "mean_token_accuracy": 0.7995368830859662,
166
+ "num_tokens": 375876.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.38624019312009655,
171
+ "eval_entropy": 0.7746140591883928,
172
+ "eval_loss": 0.7175475358963013,
173
+ "eval_mean_token_accuracy": 0.7994848955213354,
174
+ "eval_num_tokens": 375876.0,
175
+ "eval_runtime": 86.655,
176
+ "eval_samples_per_second": 16.387,
177
+ "eval_steps_per_second": 2.054,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.779565304517746,
182
+ "epoch": 0.43452021726010864,
183
+ "grad_norm": 1.032624363899231,
184
+ "learning_rate": 0.00016230198959472835,
185
+ "loss": 0.6978332042694092,
186
+ "mean_token_accuracy": 0.8015618294477462,
187
+ "num_tokens": 423913.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.43452021726010864,
192
+ "eval_entropy": 0.7455583871080634,
193
+ "eval_loss": 0.7118850350379944,
194
+ "eval_mean_token_accuracy": 0.7999682195400923,
195
+ "eval_num_tokens": 423913.0,
196
+ "eval_runtime": 86.7842,
197
+ "eval_samples_per_second": 16.362,
198
+ "eval_steps_per_second": 2.051,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.7810172818601131,
203
+ "epoch": 0.4828002414001207,
204
+ "grad_norm": 0.9276663064956665,
205
+ "learning_rate": 0.00018043629010810582,
206
+ "loss": 0.6937861442565918,
207
+ "mean_token_accuracy": 0.802893053740263,
208
+ "num_tokens": 472395.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.4828002414001207,
213
+ "eval_entropy": 0.7650193290764027,
214
+ "eval_loss": 0.7020567059516907,
215
+ "eval_mean_token_accuracy": 0.8045302153973097,
216
+ "eval_num_tokens": 472395.0,
217
+ "eval_runtime": 88.326,
218
+ "eval_samples_per_second": 16.077,
219
+ "eval_steps_per_second": 2.015,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.7862060204148292,
224
+ "epoch": 0.5310802655401328,
225
+ "grad_norm": 0.8195295929908752,
226
+ "learning_rate": 0.0001985705906214833,
227
+ "loss": 0.6878782272338867,
228
+ "mean_token_accuracy": 0.8027168907225132,
229
+ "num_tokens": 517049.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.5310802655401328,
234
+ "eval_entropy": 0.7656565583154057,
235
+ "eval_loss": 0.6973932981491089,
236
+ "eval_mean_token_accuracy": 0.8047967278555538,
237
+ "eval_num_tokens": 517049.0,
238
+ "eval_runtime": 87.7698,
239
+ "eval_samples_per_second": 16.179,
240
+ "eval_steps_per_second": 2.028,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.7611534461379051,
245
+ "epoch": 0.5793602896801449,
246
+ "grad_norm": 0.7499712109565735,
247
+ "learning_rate": 0.00021670489113486077,
248
+ "loss": 0.681937837600708,
249
+ "mean_token_accuracy": 0.8041978515684605,
250
+ "num_tokens": 562836.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.5793602896801449,
255
+ "eval_entropy": 0.7599442380197933,
256
+ "eval_loss": 0.7002046704292297,
257
+ "eval_mean_token_accuracy": 0.8051861517884759,
258
+ "eval_num_tokens": 562836.0,
259
+ "eval_runtime": 87.9239,
260
+ "eval_samples_per_second": 16.15,
261
+ "eval_steps_per_second": 2.024,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.7800783462822437,
266
+ "epoch": 0.627640313820157,
267
+ "grad_norm": 0.8645269274711609,
268
+ "learning_rate": 0.00023483919164823825,
269
+ "loss": 0.6954500675201416,
270
+ "mean_token_accuracy": 0.8031484372913837,
271
+ "num_tokens": 606612.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.627640313820157,
276
+ "eval_entropy": 0.7498679090751691,
277
+ "eval_loss": 0.7046149373054504,
278
+ "eval_mean_token_accuracy": 0.802388215667746,
279
+ "eval_num_tokens": 606612.0,
280
+ "eval_runtime": 88.2468,
281
+ "eval_samples_per_second": 16.091,
282
+ "eval_steps_per_second": 2.017,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.7586533665657044,
287
+ "epoch": 0.6759203379601689,
288
+ "grad_norm": 0.741596519947052,
289
+ "learning_rate": 0.0002529734921616157,
290
+ "loss": 0.690476369857788,
291
+ "mean_token_accuracy": 0.8046923100948333,
292
+ "num_tokens": 655920.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6759203379601689,
297
+ "eval_entropy": 0.7714274066887544,
298
+ "eval_loss": 0.6992160677909851,
299
+ "eval_mean_token_accuracy": 0.8042646482419432,
300
+ "eval_num_tokens": 655920.0,
301
+ "eval_runtime": 88.1754,
302
+ "eval_samples_per_second": 16.104,
303
+ "eval_steps_per_second": 2.019,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.7585848286747933,
308
+ "epoch": 0.724200362100181,
309
+ "grad_norm": 1.0778124332427979,
310
+ "learning_rate": 0.00027110779267499317,
311
+ "loss": 0.6790409564971924,
312
+ "mean_token_accuracy": 0.8060351841151714,
313
+ "num_tokens": 702638.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.724200362100181,
318
+ "eval_entropy": 0.7380402811457601,
319
+ "eval_loss": 0.7054820656776428,
320
+ "eval_mean_token_accuracy": 0.8003534738267406,
321
+ "eval_num_tokens": 702638.0,
322
+ "eval_runtime": 88.238,
323
+ "eval_samples_per_second": 16.093,
324
+ "eval_steps_per_second": 2.017,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.7696981988847256,
329
+ "epoch": 0.7724803862401931,
330
+ "grad_norm": 0.8408658504486084,
331
+ "learning_rate": 0.00028924209318837064,
332
+ "loss": 0.7070387363433838,
333
+ "mean_token_accuracy": 0.8035947173833847,
334
+ "num_tokens": 749377.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.7724803862401931,
339
+ "eval_entropy": 0.7897409023193831,
340
+ "eval_loss": 0.7015668749809265,
341
+ "eval_mean_token_accuracy": 0.805306687113944,
342
+ "eval_num_tokens": 749377.0,
343
+ "eval_runtime": 87.9395,
344
+ "eval_samples_per_second": 16.147,
345
+ "eval_steps_per_second": 2.024,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.762337189912796,
350
+ "epoch": 0.8207604103802052,
351
+ "grad_norm": 0.9843310117721558,
352
+ "learning_rate": 0.0003073763937017481,
353
+ "loss": 0.7021088123321533,
354
+ "mean_token_accuracy": 0.8029616877436638,
355
+ "num_tokens": 798874.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8207604103802052,
360
+ "eval_entropy": 0.7210077121016685,
361
+ "eval_loss": 0.7082746624946594,
362
+ "eval_mean_token_accuracy": 0.8049283824609906,
363
+ "eval_num_tokens": 798874.0,
364
+ "eval_runtime": 88.1706,
365
+ "eval_samples_per_second": 16.105,
366
+ "eval_steps_per_second": 2.019,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.7879025422036647,
371
+ "epoch": 0.8690404345202173,
372
+ "grad_norm": 0.9713814854621887,
373
+ "learning_rate": 0.0003255106942151256,
374
+ "loss": 0.714624547958374,
375
+ "mean_token_accuracy": 0.79987031519413,
376
+ "num_tokens": 840825.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8690404345202173,
381
+ "eval_entropy": 0.7866930489459735,
382
+ "eval_loss": 0.7145519852638245,
383
+ "eval_mean_token_accuracy": 0.8015657988157165,
384
+ "eval_num_tokens": 840825.0,
385
+ "eval_runtime": 88.3746,
386
+ "eval_samples_per_second": 16.068,
387
+ "eval_steps_per_second": 2.014,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.7844726584851742,
392
+ "epoch": 0.9173204586602294,
393
+ "grad_norm": 1.2028311491012573,
394
+ "learning_rate": 0.00034364499472850306,
395
+ "loss": 0.7040786266326904,
396
+ "mean_token_accuracy": 0.7981885217130185,
397
+ "num_tokens": 882885.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9173204586602294,
402
+ "eval_entropy": 0.7767668339643585,
403
+ "eval_loss": 0.7073574066162109,
404
+ "eval_mean_token_accuracy": 0.8040851833445303,
405
+ "eval_num_tokens": 882885.0,
406
+ "eval_runtime": 88.3828,
407
+ "eval_samples_per_second": 16.066,
408
+ "eval_steps_per_second": 2.014,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.7699790760874748,
413
+ "epoch": 0.9656004828002414,
414
+ "grad_norm": 0.9397398233413696,
415
+ "learning_rate": 0.00036177929524188054,
416
+ "loss": 0.7040313720703125,
417
+ "mean_token_accuracy": 0.8034727744758129,
418
+ "num_tokens": 927456.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9656004828002414,
423
+ "eval_entropy": 0.7749103545472863,
424
+ "eval_loss": 0.7093513011932373,
425
+ "eval_mean_token_accuracy": 0.8019482901926791,
426
+ "eval_num_tokens": 927456.0,
427
+ "eval_runtime": 86.2112,
428
+ "eval_samples_per_second": 16.471,
429
+ "eval_steps_per_second": 2.065,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.7672389172888422,
434
+ "epoch": 1.012070006035003,
435
+ "grad_norm": 0.99210524559021,
436
+ "learning_rate": 0.00037628567078152296,
437
+ "loss": 0.6913118362426758,
438
+ "mean_token_accuracy": 0.8042210944287189,
439
+ "num_tokens": 972662.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.012070006035003,
444
+ "eval_entropy": 0.7294771229283193,
445
+ "eval_loss": 0.7171520590782166,
446
+ "eval_mean_token_accuracy": 0.8015128585059991,
447
+ "eval_num_tokens": 972662.0,
448
+ "eval_runtime": 88.037,
449
+ "eval_samples_per_second": 16.13,
450
+ "eval_steps_per_second": 2.022,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.7143943183124065,
455
+ "epoch": 1.060350030175015,
456
+ "grad_norm": 1.087638258934021,
457
+ "learning_rate": 0.0003762484015601069,
458
+ "loss": 0.642704439163208,
459
+ "mean_token_accuracy": 0.8124966286122799,
460
+ "num_tokens": 1022525.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.060350030175015,
465
+ "eval_entropy": 0.743511886744017,
466
+ "eval_loss": 0.7116662859916687,
467
+ "eval_mean_token_accuracy": 0.8037210672758939,
468
+ "eval_num_tokens": 1022525.0,
469
+ "eval_runtime": 87.9931,
470
+ "eval_samples_per_second": 16.138,
471
+ "eval_steps_per_second": 2.023,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.7237346574664116,
476
+ "epoch": 1.1086300543150271,
477
+ "grad_norm": 0.9365553259849548,
478
+ "learning_rate": 0.0003761579008392123,
479
+ "loss": 0.652644681930542,
480
+ "mean_token_accuracy": 0.8107496216893196,
481
+ "num_tokens": 1069194.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1086300543150271,
486
+ "eval_entropy": 0.7372823839776972,
487
+ "eval_loss": 0.7238303422927856,
488
+ "eval_mean_token_accuracy": 0.8026230124275336,
489
+ "eval_num_tokens": 1069194.0,
490
+ "eval_runtime": 87.6778,
491
+ "eval_samples_per_second": 16.196,
492
+ "eval_steps_per_second": 2.03,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.7202137842774391,
497
+ "epoch": 1.1569100784550392,
498
+ "grad_norm": 0.7998443841934204,
499
+ "learning_rate": 0.0003760141942294726,
500
+ "loss": 0.6485219478607178,
501
+ "mean_token_accuracy": 0.8092714451253414,
502
+ "num_tokens": 1117570.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1569100784550392,
507
+ "eval_entropy": 0.71547647942318,
508
+ "eval_loss": 0.7124277949333191,
509
+ "eval_mean_token_accuracy": 0.8013174219077892,
510
+ "eval_num_tokens": 1117570.0,
511
+ "eval_runtime": 87.4387,
512
+ "eval_samples_per_second": 16.24,
513
+ "eval_steps_per_second": 2.036,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.7270819500088692,
518
+ "epoch": 1.2051901025950513,
519
+ "grad_norm": 0.9470787644386292,
520
+ "learning_rate": 0.00037581732239815854,
521
+ "loss": 0.66352219581604,
522
+ "mean_token_accuracy": 0.8102593503892421,
523
+ "num_tokens": 1161717.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.2051901025950513,
528
+ "eval_entropy": 0.7791565897759427,
529
+ "eval_loss": 0.7179726958274841,
530
+ "eval_mean_token_accuracy": 0.7948898328154275,
531
+ "eval_num_tokens": 1161717.0,
532
+ "eval_runtime": 87.9927,
533
+ "eval_samples_per_second": 16.138,
534
+ "eval_steps_per_second": 2.023,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.7320936664938926,
539
+ "epoch": 1.2534701267350634,
540
+ "grad_norm": 1.0220379829406738,
541
+ "learning_rate": 0.0003755673410576695,
542
+ "loss": 0.6581549644470215,
543
+ "mean_token_accuracy": 0.8098709337413311,
544
+ "num_tokens": 1209390.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.2534701267350634,
549
+ "eval_entropy": 0.7710678225822662,
550
+ "eval_loss": 0.708797812461853,
551
+ "eval_mean_token_accuracy": 0.8039572778042783,
552
+ "eval_num_tokens": 1209390.0,
553
+ "eval_runtime": 88.1193,
554
+ "eval_samples_per_second": 16.115,
555
+ "eval_steps_per_second": 2.02,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.7281674664467573,
560
+ "epoch": 1.3017501508750755,
561
+ "grad_norm": 0.8775188326835632,
562
+ "learning_rate": 0.000375264320949768,
563
+ "loss": 0.6631278991699219,
564
+ "mean_token_accuracy": 0.8101261422038079,
565
+ "num_tokens": 1256611.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3017501508750755,
570
+ "eval_entropy": 0.7199799800187014,
571
+ "eval_loss": 0.6981531381607056,
572
+ "eval_mean_token_accuracy": 0.8071265254127845,
573
+ "eval_num_tokens": 1256611.0,
574
+ "eval_runtime": 88.165,
575
+ "eval_samples_per_second": 16.106,
576
+ "eval_steps_per_second": 2.019,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.722860773652792,
581
+ "epoch": 1.3500301750150876,
582
+ "grad_norm": 0.8949288129806519,
583
+ "learning_rate": 0.00037490834782556,
584
+ "loss": 0.6639598846435547,
585
+ "mean_token_accuracy": 0.8105649061501026,
586
+ "num_tokens": 1302325.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3500301750150876,
591
+ "eval_entropy": 0.7134760032209118,
592
+ "eval_loss": 0.7029948234558105,
593
+ "eval_mean_token_accuracy": 0.8062985880991046,
594
+ "eval_num_tokens": 1302325.0,
595
+ "eval_runtime": 88.1286,
596
+ "eval_samples_per_second": 16.113,
597
+ "eval_steps_per_second": 2.02,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.7299126699566841,
602
+ "epoch": 1.3983101991550995,
603
+ "grad_norm": 1.0323039293289185,
604
+ "learning_rate": 0.0003744995224212291,
605
+ "loss": 0.6535763263702392,
606
+ "mean_token_accuracy": 0.8120224848389626,
607
+ "num_tokens": 1343844.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.3983101991550995,
612
+ "eval_entropy": 0.7169701182440426,
613
+ "eval_loss": 0.7067857980728149,
614
+ "eval_mean_token_accuracy": 0.8040432936689826,
615
+ "eval_num_tokens": 1343844.0,
616
+ "eval_runtime": 88.2252,
617
+ "eval_samples_per_second": 16.095,
618
+ "eval_steps_per_second": 2.018,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.7281742256134749,
623
+ "epoch": 1.4465902232951118,
624
+ "grad_norm": 1.3370424509048462,
625
+ "learning_rate": 0.00037403796042952863,
626
+ "loss": 0.6637272357940673,
627
+ "mean_token_accuracy": 0.8118261776864528,
628
+ "num_tokens": 1389471.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4465902232951118,
633
+ "eval_entropy": 0.7378843212395572,
634
+ "eval_loss": 0.7038390040397644,
635
+ "eval_mean_token_accuracy": 0.8059229371922739,
636
+ "eval_num_tokens": 1389471.0,
637
+ "eval_runtime": 87.9301,
638
+ "eval_samples_per_second": 16.149,
639
+ "eval_steps_per_second": 2.024,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.7388614989817143,
644
+ "epoch": 1.4948702474351236,
645
+ "grad_norm": 0.8464171290397644,
646
+ "learning_rate": 0.00037352379246704257,
647
+ "loss": 0.6604740142822265,
648
+ "mean_token_accuracy": 0.8102974124252796,
649
+ "num_tokens": 1435975.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.4948702474351236,
654
+ "eval_entropy": 0.7126847046814607,
655
+ "eval_loss": 0.6950703263282776,
656
+ "eval_mean_token_accuracy": 0.8085885392815879,
657
+ "eval_num_tokens": 1435975.0,
658
+ "eval_runtime": 88.2732,
659
+ "eval_samples_per_second": 16.086,
660
+ "eval_steps_per_second": 2.016,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.7258916139602661,
665
+ "epoch": 1.5431502715751357,
666
+ "grad_norm": 1.1780085563659668,
667
+ "learning_rate": 0.0003729571640372223,
668
+ "loss": 0.6529891014099121,
669
+ "mean_token_accuracy": 0.8085981778800487,
670
+ "num_tokens": 1482035.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.5431502715751357,
675
+ "eval_entropy": 0.7096509638797032,
676
+ "eval_loss": 0.690051257610321,
677
+ "eval_mean_token_accuracy": 0.8100520241796301,
678
+ "eval_num_tokens": 1482035.0,
679
+ "eval_runtime": 88.3153,
680
+ "eval_samples_per_second": 16.079,
681
+ "eval_steps_per_second": 2.016,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.7151173129677773,
686
+ "epoch": 1.5914302957151478,
687
+ "grad_norm": 1.0357667207717896,
688
+ "learning_rate": 0.0003723382354892108,
689
+ "loss": 0.6451474189758301,
690
+ "mean_token_accuracy": 0.8136902332305909,
691
+ "num_tokens": 1528119.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.5914302957151478,
696
+ "eval_entropy": 0.7060252652409371,
697
+ "eval_loss": 0.6899483799934387,
698
+ "eval_mean_token_accuracy": 0.810834194502134,
699
+ "eval_num_tokens": 1528119.0,
700
+ "eval_runtime": 87.8848,
701
+ "eval_samples_per_second": 16.158,
702
+ "eval_steps_per_second": 2.025,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.7141377851366997,
707
+ "epoch": 1.63971031985516,
708
+ "grad_norm": 0.8391667008399963,
709
+ "learning_rate": 0.000371667181972466,
710
+ "loss": 0.6481579780578614,
711
+ "mean_token_accuracy": 0.8159015104174614,
712
+ "num_tokens": 1573661.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.63971031985516,
717
+ "eval_entropy": 0.7598993319473909,
718
+ "eval_loss": 0.681204080581665,
719
+ "eval_mean_token_accuracy": 0.8100194586126992,
720
+ "eval_num_tokens": 1573661.0,
721
+ "eval_runtime": 88.2415,
722
+ "eval_samples_per_second": 16.092,
723
+ "eval_steps_per_second": 2.017,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.7235980287194252,
728
+ "epoch": 1.687990343995172,
729
+ "grad_norm": 0.8197196125984192,
730
+ "learning_rate": 0.00037094419338719537,
731
+ "loss": 0.6582849025726318,
732
+ "mean_token_accuracy": 0.8114834539592266,
733
+ "num_tokens": 1621714.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.687990343995172,
738
+ "eval_entropy": 0.7166854568411795,
739
+ "eval_loss": 0.679177463054657,
740
+ "eval_mean_token_accuracy": 0.8119039877077167,
741
+ "eval_num_tokens": 1621714.0,
742
+ "eval_runtime": 88.0957,
743
+ "eval_samples_per_second": 16.119,
744
+ "eval_steps_per_second": 2.021,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.7278190158307553,
749
+ "epoch": 1.736270368135184,
750
+ "grad_norm": 0.7820602655410767,
751
+ "learning_rate": 0.0003701694743306164,
752
+ "loss": 0.6633153438568116,
753
+ "mean_token_accuracy": 0.8123249113559723,
754
+ "num_tokens": 1672118.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.736270368135184,
759
+ "eval_entropy": 0.6980850314826108,
760
+ "eval_loss": 0.6775653958320618,
761
+ "eval_mean_token_accuracy": 0.8116872906684875,
762
+ "eval_num_tokens": 1672118.0,
763
+ "eval_runtime": 87.9795,
764
+ "eval_samples_per_second": 16.14,
765
+ "eval_steps_per_second": 2.023,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.7149631556123495,
770
+ "epoch": 1.7845503922751962,
771
+ "grad_norm": 1.0139760971069336,
772
+ "learning_rate": 0.00036934324403905795,
773
+ "loss": 0.6468976020812989,
774
+ "mean_token_accuracy": 0.8117561548948288,
775
+ "num_tokens": 1720465.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.7845503922751962,
780
+ "eval_entropy": 0.69009849094273,
781
+ "eval_loss": 0.6731093525886536,
782
+ "eval_mean_token_accuracy": 0.81281964196248,
783
+ "eval_num_tokens": 1720465.0,
784
+ "eval_runtime": 87.7412,
785
+ "eval_samples_per_second": 16.184,
786
+ "eval_steps_per_second": 2.029,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.7226049453020096,
791
+ "epoch": 1.832830416415208,
792
+ "grad_norm": 1.5040546655654907,
793
+ "learning_rate": 0.0003684657363259193,
794
+ "loss": 0.6385328769683838,
795
+ "mean_token_accuracy": 0.8132639616727829,
796
+ "num_tokens": 1764440.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.832830416415208,
801
+ "eval_entropy": 0.7125169618076153,
802
+ "eval_loss": 0.6864680647850037,
803
+ "eval_mean_token_accuracy": 0.8049418089095126,
804
+ "eval_num_tokens": 1764440.0,
805
+ "eval_runtime": 87.9429,
806
+ "eval_samples_per_second": 16.147,
807
+ "eval_steps_per_second": 2.024,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.707821810990572,
812
+ "epoch": 1.8811104405552204,
813
+ "grad_norm": 0.8604223132133484,
814
+ "learning_rate": 0.00036753719951550327,
815
+ "loss": 0.6397049427032471,
816
+ "mean_token_accuracy": 0.8137877315282822,
817
+ "num_tokens": 1810828.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.8811104405552204,
822
+ "eval_entropy": 0.6492102551326323,
823
+ "eval_loss": 0.6743726134300232,
824
+ "eval_mean_token_accuracy": 0.8130356572317273,
825
+ "eval_num_tokens": 1810828.0,
826
+ "eval_runtime": 87.8684,
827
+ "eval_samples_per_second": 16.161,
828
+ "eval_steps_per_second": 2.026,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.6800355084240437,
833
+ "epoch": 1.9293904646952322,
834
+ "grad_norm": 0.7270982265472412,
835
+ "learning_rate": 0.00036655789637274377,
836
+ "loss": 0.6267755031585693,
837
+ "mean_token_accuracy": 0.8180945307016373,
838
+ "num_tokens": 1857600.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9293904646952322,
843
+ "eval_entropy": 0.6727538651294922,
844
+ "eval_loss": 0.6696028709411621,
845
+ "eval_mean_token_accuracy": 0.8116443090224534,
846
+ "eval_num_tokens": 1857600.0,
847
+ "eval_runtime": 88.2449,
848
+ "eval_samples_per_second": 16.092,
849
+ "eval_steps_per_second": 2.017,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.6746691606938839,
854
+ "epoch": 1.9776704888352445,
855
+ "grad_norm": 0.9771540760993958,
856
+ "learning_rate": 0.0003655281040288461,
857
+ "loss": 0.6326992988586426,
858
+ "mean_token_accuracy": 0.8179976396262646,
859
+ "num_tokens": 1901998.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 1.9776704888352445,
864
+ "eval_entropy": 0.6636319716324967,
865
+ "eval_loss": 0.6714363098144531,
866
+ "eval_mean_token_accuracy": 0.8126635943235976,
867
+ "eval_num_tokens": 1901998.0,
868
+ "eval_runtime": 87.7666,
869
+ "eval_samples_per_second": 16.179,
870
+ "eval_steps_per_second": 2.028,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.6319125778489298,
875
+ "epoch": 2.024140012070006,
876
+ "grad_norm": 0.8380444645881653,
877
+ "learning_rate": 0.0003644481139028622,
878
+ "loss": 0.5780903816223144,
879
+ "mean_token_accuracy": 0.8250188401767186,
880
+ "num_tokens": 1947553.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.024140012070006,
885
+ "eval_entropy": 0.6627151989535,
886
+ "eval_loss": 0.6774935722351074,
887
+ "eval_mean_token_accuracy": 0.8140510819601209,
888
+ "eval_num_tokens": 1947553.0,
889
+ "eval_runtime": 87.8287,
890
+ "eval_samples_per_second": 16.168,
891
+ "eval_steps_per_second": 2.027,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.6059147048741579,
896
+ "epoch": 2.0724200362100182,
897
+ "grad_norm": 0.9745152592658997,
898
+ "learning_rate": 0.0003633182316192229,
899
+ "loss": 0.546702527999878,
900
+ "mean_token_accuracy": 0.8358186542987823,
901
+ "num_tokens": 1995234.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.0724200362100182,
906
+ "eval_entropy": 0.6620710384979677,
907
+ "eval_loss": 0.6769291162490845,
908
+ "eval_mean_token_accuracy": 0.8135082835561773,
909
+ "eval_num_tokens": 1995234.0,
910
+ "eval_runtime": 88.0008,
911
+ "eval_samples_per_second": 16.136,
912
+ "eval_steps_per_second": 2.023,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.6076117843389511,
917
+ "epoch": 2.12070006035003,
918
+ "grad_norm": 0.6914380192756653,
919
+ "learning_rate": 0.0003621387769212491,
920
+ "loss": 0.5324535846710206,
921
+ "mean_token_accuracy": 0.8366568259894848,
922
+ "num_tokens": 2040639.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.12070006035003,
927
+ "eval_entropy": 0.6592224213514435,
928
+ "eval_loss": 0.670383632183075,
929
+ "eval_mean_token_accuracy": 0.8162150088320957,
930
+ "eval_num_tokens": 2040639.0,
931
+ "eval_runtime": 88.324,
932
+ "eval_samples_per_second": 16.077,
933
+ "eval_steps_per_second": 2.015,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.6158072650432587,
938
+ "epoch": 2.1689800844900424,
939
+ "grad_norm": 0.9649198055267334,
940
+ "learning_rate": 0.0003609100835806688,
941
+ "loss": 0.5564568996429443,
942
+ "mean_token_accuracy": 0.8329654954373836,
943
+ "num_tokens": 2090604.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.1689800844900424,
948
+ "eval_entropy": 0.608588819758276,
949
+ "eval_loss": 0.6745102405548096,
950
+ "eval_mean_token_accuracy": 0.8155080542135774,
951
+ "eval_num_tokens": 2090604.0,
952
+ "eval_runtime": 87.52,
953
+ "eval_samples_per_second": 16.225,
954
+ "eval_steps_per_second": 2.034,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.5938734326511621,
959
+ "epoch": 2.2172601086300543,
960
+ "grad_norm": 0.6915557384490967,
961
+ "learning_rate": 0.0003596324993031632,
962
+ "loss": 0.5351875782012939,
963
+ "mean_token_accuracy": 0.8385631121695042,
964
+ "num_tokens": 2142246.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2172601086300543,
969
+ "eval_entropy": 0.6251631205001574,
970
+ "eval_loss": 0.6664518117904663,
971
+ "eval_mean_token_accuracy": 0.81711940417129,
972
+ "eval_num_tokens": 2142246.0,
973
+ "eval_runtime": 87.1792,
974
+ "eval_samples_per_second": 16.288,
975
+ "eval_steps_per_second": 2.042,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.623173613846302,
980
+ "epoch": 2.2655401327700666,
981
+ "grad_norm": 0.8366610407829285,
982
+ "learning_rate": 0.00035830638562997063,
983
+ "loss": 0.5488213539123535,
984
+ "mean_token_accuracy": 0.8321483485400677,
985
+ "num_tokens": 2183144.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.2655401327700666,
990
+ "eval_entropy": 0.6237085649136747,
991
+ "eval_loss": 0.68075031042099,
992
+ "eval_mean_token_accuracy": 0.814671738429016,
993
+ "eval_num_tokens": 2183144.0,
994
+ "eval_runtime": 88.0652,
995
+ "eval_samples_per_second": 16.124,
996
+ "eval_steps_per_second": 2.021,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.6236911740154027,
1001
+ "epoch": 2.3138201569100785,
1002
+ "grad_norm": 0.7812452912330627,
1003
+ "learning_rate": 0.00035693211783557416,
1004
+ "loss": 0.5696200847625732,
1005
+ "mean_token_accuracy": 0.82991396561265,
1006
+ "num_tokens": 2228098.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3138201569100785,
1011
+ "eval_entropy": 0.6327033426319615,
1012
+ "eval_loss": 0.6742984652519226,
1013
+ "eval_mean_token_accuracy": 0.8148086091105857,
1014
+ "eval_num_tokens": 2228098.0,
1015
+ "eval_runtime": 88.1711,
1016
+ "eval_samples_per_second": 16.105,
1017
+ "eval_steps_per_second": 2.019,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.6218426413834095,
1022
+ "epoch": 2.3621001810500903,
1023
+ "grad_norm": 0.8719790577888489,
1024
+ "learning_rate": 0.0003555100848215035,
1025
+ "loss": 0.5582651615142822,
1026
+ "mean_token_accuracy": 0.8323961742222309,
1027
+ "num_tokens": 2273062.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.3621001810500903,
1032
+ "eval_entropy": 0.6338012375858393,
1033
+ "eval_loss": 0.6685822606086731,
1034
+ "eval_mean_token_accuracy": 0.8152180300669724,
1035
+ "eval_num_tokens": 2273062.0,
1036
+ "eval_runtime": 88.101,
1037
+ "eval_samples_per_second": 16.118,
1038
+ "eval_steps_per_second": 2.02,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.6050681818276644,
1043
+ "epoch": 2.4103802051901027,
1044
+ "grad_norm": 0.9151566028594971,
1045
+ "learning_rate": 0.00035404068900628076,
1046
+ "loss": 0.559452486038208,
1047
+ "mean_token_accuracy": 0.8310242861509323,
1048
+ "num_tokens": 2319305.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.4103802051901027,
1053
+ "eval_entropy": 0.6634890969549672,
1054
+ "eval_loss": 0.6692465543746948,
1055
+ "eval_mean_token_accuracy": 0.8134353673190213,
1056
+ "eval_num_tokens": 2319305.0,
1057
+ "eval_runtime": 87.7711,
1058
+ "eval_samples_per_second": 16.178,
1059
+ "eval_steps_per_second": 2.028,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.6169968567788601,
1064
+ "epoch": 2.4586602293301145,
1065
+ "grad_norm": 0.7081006765365601,
1066
+ "learning_rate": 0.0003525243462115402,
1067
+ "loss": 0.5589694023132324,
1068
+ "mean_token_accuracy": 0.8357502184808254,
1069
+ "num_tokens": 2368592.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.4586602293301145,
1074
+ "eval_entropy": 0.6126393578695447,
1075
+ "eval_loss": 0.6706712245941162,
1076
+ "eval_mean_token_accuracy": 0.8155577500884452,
1077
+ "eval_num_tokens": 2368592.0,
1078
+ "eval_runtime": 88.1407,
1079
+ "eval_samples_per_second": 16.111,
1080
+ "eval_steps_per_second": 2.019,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.6072248566895724,
1085
+ "epoch": 2.506940253470127,
1086
+ "grad_norm": 0.6999370455741882,
1087
+ "learning_rate": 0.0003509614855443561,
1088
+ "loss": 0.5500593185424805,
1089
+ "mean_token_accuracy": 0.8337264291942119,
1090
+ "num_tokens": 2412859.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.506940253470127,
1095
+ "eval_entropy": 0.6417417074187418,
1096
+ "eval_loss": 0.6651163697242737,
1097
+ "eval_mean_token_accuracy": 0.8159113908081912,
1098
+ "eval_num_tokens": 2412859.0,
1099
+ "eval_runtime": 87.4854,
1100
+ "eval_samples_per_second": 16.231,
1101
+ "eval_steps_per_second": 2.035,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.6301592070609331,
1106
+ "epoch": 2.5552202776101387,
1107
+ "grad_norm": 0.8768910765647888,
1108
+ "learning_rate": 0.00034935254927581064,
1109
+ "loss": 0.5613903999328613,
1110
+ "mean_token_accuracy": 0.8298896946012974,
1111
+ "num_tokens": 2458201.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.5552202776101387,
1116
+ "eval_entropy": 0.6569794751285167,
1117
+ "eval_loss": 0.6668341159820557,
1118
+ "eval_mean_token_accuracy": 0.8126791795987761,
1119
+ "eval_num_tokens": 2458201.0,
1120
+ "eval_runtime": 88.0474,
1121
+ "eval_samples_per_second": 16.128,
1122
+ "eval_steps_per_second": 2.022,
1123
+ "step": 1060
1124
+ },
1125
+ {
1126
+ "entropy": 0.6069021724164486,
1127
+ "epoch": 2.603500301750151,
1128
+ "grad_norm": 0.9033508896827698,
1129
+ "learning_rate": 0.0003476979927158357,
1130
+ "loss": 0.5590654373168945,
1131
+ "mean_token_accuracy": 0.831520090252161,
1132
+ "num_tokens": 2505382.0,
1133
+ "step": 1080
1134
+ },
1135
+ {
1136
+ "epoch": 2.603500301750151,
1137
+ "eval_entropy": 0.6636380777600106,
1138
+ "eval_loss": 0.6596238017082214,
1139
+ "eval_mean_token_accuracy": 0.8177202826135614,
1140
+ "eval_num_tokens": 2505382.0,
1141
+ "eval_runtime": 88.08,
1142
+ "eval_samples_per_second": 16.122,
1143
+ "eval_steps_per_second": 2.021,
1144
+ "step": 1080
1145
+ },
1146
+ {
1147
+ "entropy": 0.6138021353632211,
1148
+ "epoch": 2.651780325890163,
1149
+ "grad_norm": 0.7345808148384094,
1150
+ "learning_rate": 0.0003459982840843664,
1151
+ "loss": 0.5586410522460937,
1152
+ "mean_token_accuracy": 0.8295876093208789,
1153
+ "num_tokens": 2552927.0,
1154
+ "step": 1100
1155
+ },
1156
+ {
1157
+ "epoch": 2.651780325890163,
1158
+ "eval_entropy": 0.5845904127600487,
1159
+ "eval_loss": 0.663780689239502,
1160
+ "eval_mean_token_accuracy": 0.8142095107710763,
1161
+ "eval_num_tokens": 2552927.0,
1162
+ "eval_runtime": 88.0083,
1163
+ "eval_samples_per_second": 16.135,
1164
+ "eval_steps_per_second": 2.023,
1165
+ "step": 1100
1166
+ }
1167
+ ],
1168
+ "logging_steps": 20,
1169
+ "max_steps": 4150,
1170
+ "num_input_tokens_seen": 0,
1171
+ "num_train_epochs": 10,
1172
+ "save_steps": 20,
1173
+ "stateful_callbacks": {
1174
+ "TrainerControl": {
1175
+ "args": {
1176
+ "should_epoch_stop": false,
1177
+ "should_evaluate": false,
1178
+ "should_log": false,
1179
+ "should_save": true,
1180
+ "should_training_stop": false
1181
+ },
1182
+ "attributes": {}
1183
+ }
1184
+ },
1185
+ "total_flos": 1.0779277426032845e+17,
1186
+ "train_batch_size": 4,
1187
+ "trial_name": null,
1188
+ "trial_params": null
1189
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1120/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1120/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.03828026524568501,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "v_proj",
34
+ "up_proj",
35
+ "down_proj",
36
+ "o_proj",
37
+ "q_proj",
38
+ "k_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1120/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1120/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1120/trainer_state.json ADDED
@@ -0,0 +1,1210 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.700060350030175,
6
+ "eval_steps": 20,
7
+ "global_step": 1120,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 2.172999547421932,
14
+ "epoch": 0.04828002414001207,
15
+ "grad_norm": 2.5071208477020264,
16
+ "learning_rate": 1.7227585487708594e-05,
17
+ "loss": 1.9347333908081055,
18
+ "mean_token_accuracy": 0.6086816020309925,
19
+ "num_tokens": 47778.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.04828002414001207,
24
+ "eval_entropy": 1.6909754791956269,
25
+ "eval_loss": 1.4728385210037231,
26
+ "eval_mean_token_accuracy": 0.6773593797442619,
27
+ "eval_num_tokens": 47778.0,
28
+ "eval_runtime": 90.1265,
29
+ "eval_samples_per_second": 15.756,
30
+ "eval_steps_per_second": 1.975,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.2014626950025558,
35
+ "epoch": 0.09656004828002414,
36
+ "grad_norm": 1.5074543952941895,
37
+ "learning_rate": 3.5361886001086065e-05,
38
+ "loss": 1.0867146492004394,
39
+ "mean_token_accuracy": 0.7285927519202232,
40
+ "num_tokens": 95981.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.09656004828002414,
45
+ "eval_entropy": 1.0304168352250302,
46
+ "eval_loss": 0.9211422204971313,
47
+ "eval_mean_token_accuracy": 0.7613122848312507,
48
+ "eval_num_tokens": 95981.0,
49
+ "eval_runtime": 88.6338,
50
+ "eval_samples_per_second": 16.021,
51
+ "eval_steps_per_second": 2.008,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.9454198732972146,
56
+ "epoch": 0.14484007242003621,
57
+ "grad_norm": 1.7578108310699463,
58
+ "learning_rate": 5.349618651446354e-05,
59
+ "loss": 0.8447297096252442,
60
+ "mean_token_accuracy": 0.7723424732685089,
61
+ "num_tokens": 142784.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.14484007242003621,
66
+ "eval_entropy": 0.8647834474451086,
67
+ "eval_loss": 0.8240737318992615,
68
+ "eval_mean_token_accuracy": 0.7821227014064789,
69
+ "eval_num_tokens": 142784.0,
70
+ "eval_runtime": 88.0135,
71
+ "eval_samples_per_second": 16.134,
72
+ "eval_steps_per_second": 2.022,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.8963550612330436,
77
+ "epoch": 0.19312009656004828,
78
+ "grad_norm": 1.1464931964874268,
79
+ "learning_rate": 7.1630487027841e-05,
80
+ "loss": 0.7979836463928223,
81
+ "mean_token_accuracy": 0.7822489373385906,
82
+ "num_tokens": 187338.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.19312009656004828,
87
+ "eval_entropy": 0.8663296133614657,
88
+ "eval_loss": 0.7752988338470459,
89
+ "eval_mean_token_accuracy": 0.7885957012015782,
90
+ "eval_num_tokens": 187338.0,
91
+ "eval_runtime": 88.284,
92
+ "eval_samples_per_second": 16.084,
93
+ "eval_steps_per_second": 2.016,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.8265103206038475,
98
+ "epoch": 0.24140012070006034,
99
+ "grad_norm": 1.0106662511825562,
100
+ "learning_rate": 8.976478754121848e-05,
101
+ "loss": 0.7297060012817382,
102
+ "mean_token_accuracy": 0.7955138415098191,
103
+ "num_tokens": 236038.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24140012070006034,
108
+ "eval_entropy": 0.8122169536151244,
109
+ "eval_loss": 0.754906177520752,
110
+ "eval_mean_token_accuracy": 0.7917155255092664,
111
+ "eval_num_tokens": 236038.0,
112
+ "eval_runtime": 87.5754,
113
+ "eval_samples_per_second": 16.215,
114
+ "eval_steps_per_second": 2.033,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.8098392844200134,
119
+ "epoch": 0.28968014484007243,
120
+ "grad_norm": 1.2238041162490845,
121
+ "learning_rate": 0.00010789908805459595,
122
+ "loss": 0.7201489448547364,
123
+ "mean_token_accuracy": 0.7975016921758652,
124
+ "num_tokens": 282918.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.28968014484007243,
129
+ "eval_entropy": 0.8418726462326692,
130
+ "eval_loss": 0.7393178939819336,
131
+ "eval_mean_token_accuracy": 0.7936771585700217,
132
+ "eval_num_tokens": 282918.0,
133
+ "eval_runtime": 87.1597,
134
+ "eval_samples_per_second": 16.292,
135
+ "eval_steps_per_second": 2.042,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.8080591082572937,
140
+ "epoch": 0.33796016898008446,
141
+ "grad_norm": 1.011591911315918,
142
+ "learning_rate": 0.00012603338856797343,
143
+ "loss": 0.7147697925567627,
144
+ "mean_token_accuracy": 0.7947175532579422,
145
+ "num_tokens": 331305.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.33796016898008446,
150
+ "eval_entropy": 0.7894941431083037,
151
+ "eval_loss": 0.720219075679779,
152
+ "eval_mean_token_accuracy": 0.801899236239744,
153
+ "eval_num_tokens": 331305.0,
154
+ "eval_runtime": 87.1493,
155
+ "eval_samples_per_second": 16.294,
156
+ "eval_steps_per_second": 2.042,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.7957205109298229,
161
+ "epoch": 0.38624019312009655,
162
+ "grad_norm": 0.8481830954551697,
163
+ "learning_rate": 0.00014416768908135088,
164
+ "loss": 0.7050027847290039,
165
+ "mean_token_accuracy": 0.7995368830859662,
166
+ "num_tokens": 375876.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.38624019312009655,
171
+ "eval_entropy": 0.7746140591883928,
172
+ "eval_loss": 0.7175475358963013,
173
+ "eval_mean_token_accuracy": 0.7994848955213354,
174
+ "eval_num_tokens": 375876.0,
175
+ "eval_runtime": 86.655,
176
+ "eval_samples_per_second": 16.387,
177
+ "eval_steps_per_second": 2.054,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.779565304517746,
182
+ "epoch": 0.43452021726010864,
183
+ "grad_norm": 1.032624363899231,
184
+ "learning_rate": 0.00016230198959472835,
185
+ "loss": 0.6978332042694092,
186
+ "mean_token_accuracy": 0.8015618294477462,
187
+ "num_tokens": 423913.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.43452021726010864,
192
+ "eval_entropy": 0.7455583871080634,
193
+ "eval_loss": 0.7118850350379944,
194
+ "eval_mean_token_accuracy": 0.7999682195400923,
195
+ "eval_num_tokens": 423913.0,
196
+ "eval_runtime": 86.7842,
197
+ "eval_samples_per_second": 16.362,
198
+ "eval_steps_per_second": 2.051,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.7810172818601131,
203
+ "epoch": 0.4828002414001207,
204
+ "grad_norm": 0.9276663064956665,
205
+ "learning_rate": 0.00018043629010810582,
206
+ "loss": 0.6937861442565918,
207
+ "mean_token_accuracy": 0.802893053740263,
208
+ "num_tokens": 472395.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.4828002414001207,
213
+ "eval_entropy": 0.7650193290764027,
214
+ "eval_loss": 0.7020567059516907,
215
+ "eval_mean_token_accuracy": 0.8045302153973097,
216
+ "eval_num_tokens": 472395.0,
217
+ "eval_runtime": 88.326,
218
+ "eval_samples_per_second": 16.077,
219
+ "eval_steps_per_second": 2.015,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.7862060204148292,
224
+ "epoch": 0.5310802655401328,
225
+ "grad_norm": 0.8195295929908752,
226
+ "learning_rate": 0.0001985705906214833,
227
+ "loss": 0.6878782272338867,
228
+ "mean_token_accuracy": 0.8027168907225132,
229
+ "num_tokens": 517049.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.5310802655401328,
234
+ "eval_entropy": 0.7656565583154057,
235
+ "eval_loss": 0.6973932981491089,
236
+ "eval_mean_token_accuracy": 0.8047967278555538,
237
+ "eval_num_tokens": 517049.0,
238
+ "eval_runtime": 87.7698,
239
+ "eval_samples_per_second": 16.179,
240
+ "eval_steps_per_second": 2.028,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.7611534461379051,
245
+ "epoch": 0.5793602896801449,
246
+ "grad_norm": 0.7499712109565735,
247
+ "learning_rate": 0.00021670489113486077,
248
+ "loss": 0.681937837600708,
249
+ "mean_token_accuracy": 0.8041978515684605,
250
+ "num_tokens": 562836.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.5793602896801449,
255
+ "eval_entropy": 0.7599442380197933,
256
+ "eval_loss": 0.7002046704292297,
257
+ "eval_mean_token_accuracy": 0.8051861517884759,
258
+ "eval_num_tokens": 562836.0,
259
+ "eval_runtime": 87.9239,
260
+ "eval_samples_per_second": 16.15,
261
+ "eval_steps_per_second": 2.024,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.7800783462822437,
266
+ "epoch": 0.627640313820157,
267
+ "grad_norm": 0.8645269274711609,
268
+ "learning_rate": 0.00023483919164823825,
269
+ "loss": 0.6954500675201416,
270
+ "mean_token_accuracy": 0.8031484372913837,
271
+ "num_tokens": 606612.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.627640313820157,
276
+ "eval_entropy": 0.7498679090751691,
277
+ "eval_loss": 0.7046149373054504,
278
+ "eval_mean_token_accuracy": 0.802388215667746,
279
+ "eval_num_tokens": 606612.0,
280
+ "eval_runtime": 88.2468,
281
+ "eval_samples_per_second": 16.091,
282
+ "eval_steps_per_second": 2.017,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.7586533665657044,
287
+ "epoch": 0.6759203379601689,
288
+ "grad_norm": 0.741596519947052,
289
+ "learning_rate": 0.0002529734921616157,
290
+ "loss": 0.690476369857788,
291
+ "mean_token_accuracy": 0.8046923100948333,
292
+ "num_tokens": 655920.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6759203379601689,
297
+ "eval_entropy": 0.7714274066887544,
298
+ "eval_loss": 0.6992160677909851,
299
+ "eval_mean_token_accuracy": 0.8042646482419432,
300
+ "eval_num_tokens": 655920.0,
301
+ "eval_runtime": 88.1754,
302
+ "eval_samples_per_second": 16.104,
303
+ "eval_steps_per_second": 2.019,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.7585848286747933,
308
+ "epoch": 0.724200362100181,
309
+ "grad_norm": 1.0778124332427979,
310
+ "learning_rate": 0.00027110779267499317,
311
+ "loss": 0.6790409564971924,
312
+ "mean_token_accuracy": 0.8060351841151714,
313
+ "num_tokens": 702638.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.724200362100181,
318
+ "eval_entropy": 0.7380402811457601,
319
+ "eval_loss": 0.7054820656776428,
320
+ "eval_mean_token_accuracy": 0.8003534738267406,
321
+ "eval_num_tokens": 702638.0,
322
+ "eval_runtime": 88.238,
323
+ "eval_samples_per_second": 16.093,
324
+ "eval_steps_per_second": 2.017,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.7696981988847256,
329
+ "epoch": 0.7724803862401931,
330
+ "grad_norm": 0.8408658504486084,
331
+ "learning_rate": 0.00028924209318837064,
332
+ "loss": 0.7070387363433838,
333
+ "mean_token_accuracy": 0.8035947173833847,
334
+ "num_tokens": 749377.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.7724803862401931,
339
+ "eval_entropy": 0.7897409023193831,
340
+ "eval_loss": 0.7015668749809265,
341
+ "eval_mean_token_accuracy": 0.805306687113944,
342
+ "eval_num_tokens": 749377.0,
343
+ "eval_runtime": 87.9395,
344
+ "eval_samples_per_second": 16.147,
345
+ "eval_steps_per_second": 2.024,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.762337189912796,
350
+ "epoch": 0.8207604103802052,
351
+ "grad_norm": 0.9843310117721558,
352
+ "learning_rate": 0.0003073763937017481,
353
+ "loss": 0.7021088123321533,
354
+ "mean_token_accuracy": 0.8029616877436638,
355
+ "num_tokens": 798874.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8207604103802052,
360
+ "eval_entropy": 0.7210077121016685,
361
+ "eval_loss": 0.7082746624946594,
362
+ "eval_mean_token_accuracy": 0.8049283824609906,
363
+ "eval_num_tokens": 798874.0,
364
+ "eval_runtime": 88.1706,
365
+ "eval_samples_per_second": 16.105,
366
+ "eval_steps_per_second": 2.019,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.7879025422036647,
371
+ "epoch": 0.8690404345202173,
372
+ "grad_norm": 0.9713814854621887,
373
+ "learning_rate": 0.0003255106942151256,
374
+ "loss": 0.714624547958374,
375
+ "mean_token_accuracy": 0.79987031519413,
376
+ "num_tokens": 840825.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8690404345202173,
381
+ "eval_entropy": 0.7866930489459735,
382
+ "eval_loss": 0.7145519852638245,
383
+ "eval_mean_token_accuracy": 0.8015657988157165,
384
+ "eval_num_tokens": 840825.0,
385
+ "eval_runtime": 88.3746,
386
+ "eval_samples_per_second": 16.068,
387
+ "eval_steps_per_second": 2.014,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.7844726584851742,
392
+ "epoch": 0.9173204586602294,
393
+ "grad_norm": 1.2028311491012573,
394
+ "learning_rate": 0.00034364499472850306,
395
+ "loss": 0.7040786266326904,
396
+ "mean_token_accuracy": 0.7981885217130185,
397
+ "num_tokens": 882885.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9173204586602294,
402
+ "eval_entropy": 0.7767668339643585,
403
+ "eval_loss": 0.7073574066162109,
404
+ "eval_mean_token_accuracy": 0.8040851833445303,
405
+ "eval_num_tokens": 882885.0,
406
+ "eval_runtime": 88.3828,
407
+ "eval_samples_per_second": 16.066,
408
+ "eval_steps_per_second": 2.014,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.7699790760874748,
413
+ "epoch": 0.9656004828002414,
414
+ "grad_norm": 0.9397398233413696,
415
+ "learning_rate": 0.00036177929524188054,
416
+ "loss": 0.7040313720703125,
417
+ "mean_token_accuracy": 0.8034727744758129,
418
+ "num_tokens": 927456.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9656004828002414,
423
+ "eval_entropy": 0.7749103545472863,
424
+ "eval_loss": 0.7093513011932373,
425
+ "eval_mean_token_accuracy": 0.8019482901926791,
426
+ "eval_num_tokens": 927456.0,
427
+ "eval_runtime": 86.2112,
428
+ "eval_samples_per_second": 16.471,
429
+ "eval_steps_per_second": 2.065,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.7672389172888422,
434
+ "epoch": 1.012070006035003,
435
+ "grad_norm": 0.99210524559021,
436
+ "learning_rate": 0.00037628567078152296,
437
+ "loss": 0.6913118362426758,
438
+ "mean_token_accuracy": 0.8042210944287189,
439
+ "num_tokens": 972662.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.012070006035003,
444
+ "eval_entropy": 0.7294771229283193,
445
+ "eval_loss": 0.7171520590782166,
446
+ "eval_mean_token_accuracy": 0.8015128585059991,
447
+ "eval_num_tokens": 972662.0,
448
+ "eval_runtime": 88.037,
449
+ "eval_samples_per_second": 16.13,
450
+ "eval_steps_per_second": 2.022,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.7143943183124065,
455
+ "epoch": 1.060350030175015,
456
+ "grad_norm": 1.087638258934021,
457
+ "learning_rate": 0.0003762484015601069,
458
+ "loss": 0.642704439163208,
459
+ "mean_token_accuracy": 0.8124966286122799,
460
+ "num_tokens": 1022525.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.060350030175015,
465
+ "eval_entropy": 0.743511886744017,
466
+ "eval_loss": 0.7116662859916687,
467
+ "eval_mean_token_accuracy": 0.8037210672758939,
468
+ "eval_num_tokens": 1022525.0,
469
+ "eval_runtime": 87.9931,
470
+ "eval_samples_per_second": 16.138,
471
+ "eval_steps_per_second": 2.023,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.7237346574664116,
476
+ "epoch": 1.1086300543150271,
477
+ "grad_norm": 0.9365553259849548,
478
+ "learning_rate": 0.0003761579008392123,
479
+ "loss": 0.652644681930542,
480
+ "mean_token_accuracy": 0.8107496216893196,
481
+ "num_tokens": 1069194.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1086300543150271,
486
+ "eval_entropy": 0.7372823839776972,
487
+ "eval_loss": 0.7238303422927856,
488
+ "eval_mean_token_accuracy": 0.8026230124275336,
489
+ "eval_num_tokens": 1069194.0,
490
+ "eval_runtime": 87.6778,
491
+ "eval_samples_per_second": 16.196,
492
+ "eval_steps_per_second": 2.03,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.7202137842774391,
497
+ "epoch": 1.1569100784550392,
498
+ "grad_norm": 0.7998443841934204,
499
+ "learning_rate": 0.0003760141942294726,
500
+ "loss": 0.6485219478607178,
501
+ "mean_token_accuracy": 0.8092714451253414,
502
+ "num_tokens": 1117570.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1569100784550392,
507
+ "eval_entropy": 0.71547647942318,
508
+ "eval_loss": 0.7124277949333191,
509
+ "eval_mean_token_accuracy": 0.8013174219077892,
510
+ "eval_num_tokens": 1117570.0,
511
+ "eval_runtime": 87.4387,
512
+ "eval_samples_per_second": 16.24,
513
+ "eval_steps_per_second": 2.036,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.7270819500088692,
518
+ "epoch": 1.2051901025950513,
519
+ "grad_norm": 0.9470787644386292,
520
+ "learning_rate": 0.00037581732239815854,
521
+ "loss": 0.66352219581604,
522
+ "mean_token_accuracy": 0.8102593503892421,
523
+ "num_tokens": 1161717.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.2051901025950513,
528
+ "eval_entropy": 0.7791565897759427,
529
+ "eval_loss": 0.7179726958274841,
530
+ "eval_mean_token_accuracy": 0.7948898328154275,
531
+ "eval_num_tokens": 1161717.0,
532
+ "eval_runtime": 87.9927,
533
+ "eval_samples_per_second": 16.138,
534
+ "eval_steps_per_second": 2.023,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.7320936664938926,
539
+ "epoch": 1.2534701267350634,
540
+ "grad_norm": 1.0220379829406738,
541
+ "learning_rate": 0.0003755673410576695,
542
+ "loss": 0.6581549644470215,
543
+ "mean_token_accuracy": 0.8098709337413311,
544
+ "num_tokens": 1209390.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.2534701267350634,
549
+ "eval_entropy": 0.7710678225822662,
550
+ "eval_loss": 0.708797812461853,
551
+ "eval_mean_token_accuracy": 0.8039572778042783,
552
+ "eval_num_tokens": 1209390.0,
553
+ "eval_runtime": 88.1193,
554
+ "eval_samples_per_second": 16.115,
555
+ "eval_steps_per_second": 2.02,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.7281674664467573,
560
+ "epoch": 1.3017501508750755,
561
+ "grad_norm": 0.8775188326835632,
562
+ "learning_rate": 0.000375264320949768,
563
+ "loss": 0.6631278991699219,
564
+ "mean_token_accuracy": 0.8101261422038079,
565
+ "num_tokens": 1256611.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3017501508750755,
570
+ "eval_entropy": 0.7199799800187014,
571
+ "eval_loss": 0.6981531381607056,
572
+ "eval_mean_token_accuracy": 0.8071265254127845,
573
+ "eval_num_tokens": 1256611.0,
574
+ "eval_runtime": 88.165,
575
+ "eval_samples_per_second": 16.106,
576
+ "eval_steps_per_second": 2.019,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.722860773652792,
581
+ "epoch": 1.3500301750150876,
582
+ "grad_norm": 0.8949288129806519,
583
+ "learning_rate": 0.00037490834782556,
584
+ "loss": 0.6639598846435547,
585
+ "mean_token_accuracy": 0.8105649061501026,
586
+ "num_tokens": 1302325.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3500301750150876,
591
+ "eval_entropy": 0.7134760032209118,
592
+ "eval_loss": 0.7029948234558105,
593
+ "eval_mean_token_accuracy": 0.8062985880991046,
594
+ "eval_num_tokens": 1302325.0,
595
+ "eval_runtime": 88.1286,
596
+ "eval_samples_per_second": 16.113,
597
+ "eval_steps_per_second": 2.02,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.7299126699566841,
602
+ "epoch": 1.3983101991550995,
603
+ "grad_norm": 1.0323039293289185,
604
+ "learning_rate": 0.0003744995224212291,
605
+ "loss": 0.6535763263702392,
606
+ "mean_token_accuracy": 0.8120224848389626,
607
+ "num_tokens": 1343844.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.3983101991550995,
612
+ "eval_entropy": 0.7169701182440426,
613
+ "eval_loss": 0.7067857980728149,
614
+ "eval_mean_token_accuracy": 0.8040432936689826,
615
+ "eval_num_tokens": 1343844.0,
616
+ "eval_runtime": 88.2252,
617
+ "eval_samples_per_second": 16.095,
618
+ "eval_steps_per_second": 2.018,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.7281742256134749,
623
+ "epoch": 1.4465902232951118,
624
+ "grad_norm": 1.3370424509048462,
625
+ "learning_rate": 0.00037403796042952863,
626
+ "loss": 0.6637272357940673,
627
+ "mean_token_accuracy": 0.8118261776864528,
628
+ "num_tokens": 1389471.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4465902232951118,
633
+ "eval_entropy": 0.7378843212395572,
634
+ "eval_loss": 0.7038390040397644,
635
+ "eval_mean_token_accuracy": 0.8059229371922739,
636
+ "eval_num_tokens": 1389471.0,
637
+ "eval_runtime": 87.9301,
638
+ "eval_samples_per_second": 16.149,
639
+ "eval_steps_per_second": 2.024,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.7388614989817143,
644
+ "epoch": 1.4948702474351236,
645
+ "grad_norm": 0.8464171290397644,
646
+ "learning_rate": 0.00037352379246704257,
647
+ "loss": 0.6604740142822265,
648
+ "mean_token_accuracy": 0.8102974124252796,
649
+ "num_tokens": 1435975.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.4948702474351236,
654
+ "eval_entropy": 0.7126847046814607,
655
+ "eval_loss": 0.6950703263282776,
656
+ "eval_mean_token_accuracy": 0.8085885392815879,
657
+ "eval_num_tokens": 1435975.0,
658
+ "eval_runtime": 88.2732,
659
+ "eval_samples_per_second": 16.086,
660
+ "eval_steps_per_second": 2.016,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.7258916139602661,
665
+ "epoch": 1.5431502715751357,
666
+ "grad_norm": 1.1780085563659668,
667
+ "learning_rate": 0.0003729571640372223,
668
+ "loss": 0.6529891014099121,
669
+ "mean_token_accuracy": 0.8085981778800487,
670
+ "num_tokens": 1482035.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.5431502715751357,
675
+ "eval_entropy": 0.7096509638797032,
676
+ "eval_loss": 0.690051257610321,
677
+ "eval_mean_token_accuracy": 0.8100520241796301,
678
+ "eval_num_tokens": 1482035.0,
679
+ "eval_runtime": 88.3153,
680
+ "eval_samples_per_second": 16.079,
681
+ "eval_steps_per_second": 2.016,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.7151173129677773,
686
+ "epoch": 1.5914302957151478,
687
+ "grad_norm": 1.0357667207717896,
688
+ "learning_rate": 0.0003723382354892108,
689
+ "loss": 0.6451474189758301,
690
+ "mean_token_accuracy": 0.8136902332305909,
691
+ "num_tokens": 1528119.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.5914302957151478,
696
+ "eval_entropy": 0.7060252652409371,
697
+ "eval_loss": 0.6899483799934387,
698
+ "eval_mean_token_accuracy": 0.810834194502134,
699
+ "eval_num_tokens": 1528119.0,
700
+ "eval_runtime": 87.8848,
701
+ "eval_samples_per_second": 16.158,
702
+ "eval_steps_per_second": 2.025,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.7141377851366997,
707
+ "epoch": 1.63971031985516,
708
+ "grad_norm": 0.8391667008399963,
709
+ "learning_rate": 0.000371667181972466,
710
+ "loss": 0.6481579780578614,
711
+ "mean_token_accuracy": 0.8159015104174614,
712
+ "num_tokens": 1573661.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.63971031985516,
717
+ "eval_entropy": 0.7598993319473909,
718
+ "eval_loss": 0.681204080581665,
719
+ "eval_mean_token_accuracy": 0.8100194586126992,
720
+ "eval_num_tokens": 1573661.0,
721
+ "eval_runtime": 88.2415,
722
+ "eval_samples_per_second": 16.092,
723
+ "eval_steps_per_second": 2.017,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.7235980287194252,
728
+ "epoch": 1.687990343995172,
729
+ "grad_norm": 0.8197196125984192,
730
+ "learning_rate": 0.00037094419338719537,
731
+ "loss": 0.6582849025726318,
732
+ "mean_token_accuracy": 0.8114834539592266,
733
+ "num_tokens": 1621714.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.687990343995172,
738
+ "eval_entropy": 0.7166854568411795,
739
+ "eval_loss": 0.679177463054657,
740
+ "eval_mean_token_accuracy": 0.8119039877077167,
741
+ "eval_num_tokens": 1621714.0,
742
+ "eval_runtime": 88.0957,
743
+ "eval_samples_per_second": 16.119,
744
+ "eval_steps_per_second": 2.021,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.7278190158307553,
749
+ "epoch": 1.736270368135184,
750
+ "grad_norm": 0.7820602655410767,
751
+ "learning_rate": 0.0003701694743306164,
752
+ "loss": 0.6633153438568116,
753
+ "mean_token_accuracy": 0.8123249113559723,
754
+ "num_tokens": 1672118.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.736270368135184,
759
+ "eval_entropy": 0.6980850314826108,
760
+ "eval_loss": 0.6775653958320618,
761
+ "eval_mean_token_accuracy": 0.8116872906684875,
762
+ "eval_num_tokens": 1672118.0,
763
+ "eval_runtime": 87.9795,
764
+ "eval_samples_per_second": 16.14,
765
+ "eval_steps_per_second": 2.023,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.7149631556123495,
770
+ "epoch": 1.7845503922751962,
771
+ "grad_norm": 1.0139760971069336,
772
+ "learning_rate": 0.00036934324403905795,
773
+ "loss": 0.6468976020812989,
774
+ "mean_token_accuracy": 0.8117561548948288,
775
+ "num_tokens": 1720465.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.7845503922751962,
780
+ "eval_entropy": 0.69009849094273,
781
+ "eval_loss": 0.6731093525886536,
782
+ "eval_mean_token_accuracy": 0.81281964196248,
783
+ "eval_num_tokens": 1720465.0,
784
+ "eval_runtime": 87.7412,
785
+ "eval_samples_per_second": 16.184,
786
+ "eval_steps_per_second": 2.029,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.7226049453020096,
791
+ "epoch": 1.832830416415208,
792
+ "grad_norm": 1.5040546655654907,
793
+ "learning_rate": 0.0003684657363259193,
794
+ "loss": 0.6385328769683838,
795
+ "mean_token_accuracy": 0.8132639616727829,
796
+ "num_tokens": 1764440.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.832830416415208,
801
+ "eval_entropy": 0.7125169618076153,
802
+ "eval_loss": 0.6864680647850037,
803
+ "eval_mean_token_accuracy": 0.8049418089095126,
804
+ "eval_num_tokens": 1764440.0,
805
+ "eval_runtime": 87.9429,
806
+ "eval_samples_per_second": 16.147,
807
+ "eval_steps_per_second": 2.024,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.707821810990572,
812
+ "epoch": 1.8811104405552204,
813
+ "grad_norm": 0.8604223132133484,
814
+ "learning_rate": 0.00036753719951550327,
815
+ "loss": 0.6397049427032471,
816
+ "mean_token_accuracy": 0.8137877315282822,
817
+ "num_tokens": 1810828.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.8811104405552204,
822
+ "eval_entropy": 0.6492102551326323,
823
+ "eval_loss": 0.6743726134300232,
824
+ "eval_mean_token_accuracy": 0.8130356572317273,
825
+ "eval_num_tokens": 1810828.0,
826
+ "eval_runtime": 87.8684,
827
+ "eval_samples_per_second": 16.161,
828
+ "eval_steps_per_second": 2.026,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.6800355084240437,
833
+ "epoch": 1.9293904646952322,
834
+ "grad_norm": 0.7270982265472412,
835
+ "learning_rate": 0.00036655789637274377,
836
+ "loss": 0.6267755031585693,
837
+ "mean_token_accuracy": 0.8180945307016373,
838
+ "num_tokens": 1857600.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9293904646952322,
843
+ "eval_entropy": 0.6727538651294922,
844
+ "eval_loss": 0.6696028709411621,
845
+ "eval_mean_token_accuracy": 0.8116443090224534,
846
+ "eval_num_tokens": 1857600.0,
847
+ "eval_runtime": 88.2449,
848
+ "eval_samples_per_second": 16.092,
849
+ "eval_steps_per_second": 2.017,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.6746691606938839,
854
+ "epoch": 1.9776704888352445,
855
+ "grad_norm": 0.9771540760993958,
856
+ "learning_rate": 0.0003655281040288461,
857
+ "loss": 0.6326992988586426,
858
+ "mean_token_accuracy": 0.8179976396262646,
859
+ "num_tokens": 1901998.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 1.9776704888352445,
864
+ "eval_entropy": 0.6636319716324967,
865
+ "eval_loss": 0.6714363098144531,
866
+ "eval_mean_token_accuracy": 0.8126635943235976,
867
+ "eval_num_tokens": 1901998.0,
868
+ "eval_runtime": 87.7666,
869
+ "eval_samples_per_second": 16.179,
870
+ "eval_steps_per_second": 2.028,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.6319125778489298,
875
+ "epoch": 2.024140012070006,
876
+ "grad_norm": 0.8380444645881653,
877
+ "learning_rate": 0.0003644481139028622,
878
+ "loss": 0.5780903816223144,
879
+ "mean_token_accuracy": 0.8250188401767186,
880
+ "num_tokens": 1947553.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.024140012070006,
885
+ "eval_entropy": 0.6627151989535,
886
+ "eval_loss": 0.6774935722351074,
887
+ "eval_mean_token_accuracy": 0.8140510819601209,
888
+ "eval_num_tokens": 1947553.0,
889
+ "eval_runtime": 87.8287,
890
+ "eval_samples_per_second": 16.168,
891
+ "eval_steps_per_second": 2.027,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.6059147048741579,
896
+ "epoch": 2.0724200362100182,
897
+ "grad_norm": 0.9745152592658997,
898
+ "learning_rate": 0.0003633182316192229,
899
+ "loss": 0.546702527999878,
900
+ "mean_token_accuracy": 0.8358186542987823,
901
+ "num_tokens": 1995234.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.0724200362100182,
906
+ "eval_entropy": 0.6620710384979677,
907
+ "eval_loss": 0.6769291162490845,
908
+ "eval_mean_token_accuracy": 0.8135082835561773,
909
+ "eval_num_tokens": 1995234.0,
910
+ "eval_runtime": 88.0008,
911
+ "eval_samples_per_second": 16.136,
912
+ "eval_steps_per_second": 2.023,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.6076117843389511,
917
+ "epoch": 2.12070006035003,
918
+ "grad_norm": 0.6914380192756653,
919
+ "learning_rate": 0.0003621387769212491,
920
+ "loss": 0.5324535846710206,
921
+ "mean_token_accuracy": 0.8366568259894848,
922
+ "num_tokens": 2040639.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.12070006035003,
927
+ "eval_entropy": 0.6592224213514435,
928
+ "eval_loss": 0.670383632183075,
929
+ "eval_mean_token_accuracy": 0.8162150088320957,
930
+ "eval_num_tokens": 2040639.0,
931
+ "eval_runtime": 88.324,
932
+ "eval_samples_per_second": 16.077,
933
+ "eval_steps_per_second": 2.015,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.6158072650432587,
938
+ "epoch": 2.1689800844900424,
939
+ "grad_norm": 0.9649198055267334,
940
+ "learning_rate": 0.0003609100835806688,
941
+ "loss": 0.5564568996429443,
942
+ "mean_token_accuracy": 0.8329654954373836,
943
+ "num_tokens": 2090604.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.1689800844900424,
948
+ "eval_entropy": 0.608588819758276,
949
+ "eval_loss": 0.6745102405548096,
950
+ "eval_mean_token_accuracy": 0.8155080542135774,
951
+ "eval_num_tokens": 2090604.0,
952
+ "eval_runtime": 87.52,
953
+ "eval_samples_per_second": 16.225,
954
+ "eval_steps_per_second": 2.034,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.5938734326511621,
959
+ "epoch": 2.2172601086300543,
960
+ "grad_norm": 0.6915557384490967,
961
+ "learning_rate": 0.0003596324993031632,
962
+ "loss": 0.5351875782012939,
963
+ "mean_token_accuracy": 0.8385631121695042,
964
+ "num_tokens": 2142246.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2172601086300543,
969
+ "eval_entropy": 0.6251631205001574,
970
+ "eval_loss": 0.6664518117904663,
971
+ "eval_mean_token_accuracy": 0.81711940417129,
972
+ "eval_num_tokens": 2142246.0,
973
+ "eval_runtime": 87.1792,
974
+ "eval_samples_per_second": 16.288,
975
+ "eval_steps_per_second": 2.042,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.623173613846302,
980
+ "epoch": 2.2655401327700666,
981
+ "grad_norm": 0.8366610407829285,
982
+ "learning_rate": 0.00035830638562997063,
983
+ "loss": 0.5488213539123535,
984
+ "mean_token_accuracy": 0.8321483485400677,
985
+ "num_tokens": 2183144.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.2655401327700666,
990
+ "eval_entropy": 0.6237085649136747,
991
+ "eval_loss": 0.68075031042099,
992
+ "eval_mean_token_accuracy": 0.814671738429016,
993
+ "eval_num_tokens": 2183144.0,
994
+ "eval_runtime": 88.0652,
995
+ "eval_samples_per_second": 16.124,
996
+ "eval_steps_per_second": 2.021,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.6236911740154027,
1001
+ "epoch": 2.3138201569100785,
1002
+ "grad_norm": 0.7812452912330627,
1003
+ "learning_rate": 0.00035693211783557416,
1004
+ "loss": 0.5696200847625732,
1005
+ "mean_token_accuracy": 0.82991396561265,
1006
+ "num_tokens": 2228098.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3138201569100785,
1011
+ "eval_entropy": 0.6327033426319615,
1012
+ "eval_loss": 0.6742984652519226,
1013
+ "eval_mean_token_accuracy": 0.8148086091105857,
1014
+ "eval_num_tokens": 2228098.0,
1015
+ "eval_runtime": 88.1711,
1016
+ "eval_samples_per_second": 16.105,
1017
+ "eval_steps_per_second": 2.019,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.6218426413834095,
1022
+ "epoch": 2.3621001810500903,
1023
+ "grad_norm": 0.8719790577888489,
1024
+ "learning_rate": 0.0003555100848215035,
1025
+ "loss": 0.5582651615142822,
1026
+ "mean_token_accuracy": 0.8323961742222309,
1027
+ "num_tokens": 2273062.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.3621001810500903,
1032
+ "eval_entropy": 0.6338012375858393,
1033
+ "eval_loss": 0.6685822606086731,
1034
+ "eval_mean_token_accuracy": 0.8152180300669724,
1035
+ "eval_num_tokens": 2273062.0,
1036
+ "eval_runtime": 88.101,
1037
+ "eval_samples_per_second": 16.118,
1038
+ "eval_steps_per_second": 2.02,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.6050681818276644,
1043
+ "epoch": 2.4103802051901027,
1044
+ "grad_norm": 0.9151566028594971,
1045
+ "learning_rate": 0.00035404068900628076,
1046
+ "loss": 0.559452486038208,
1047
+ "mean_token_accuracy": 0.8310242861509323,
1048
+ "num_tokens": 2319305.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.4103802051901027,
1053
+ "eval_entropy": 0.6634890969549672,
1054
+ "eval_loss": 0.6692465543746948,
1055
+ "eval_mean_token_accuracy": 0.8134353673190213,
1056
+ "eval_num_tokens": 2319305.0,
1057
+ "eval_runtime": 87.7711,
1058
+ "eval_samples_per_second": 16.178,
1059
+ "eval_steps_per_second": 2.028,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.6169968567788601,
1064
+ "epoch": 2.4586602293301145,
1065
+ "grad_norm": 0.7081006765365601,
1066
+ "learning_rate": 0.0003525243462115402,
1067
+ "loss": 0.5589694023132324,
1068
+ "mean_token_accuracy": 0.8357502184808254,
1069
+ "num_tokens": 2368592.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.4586602293301145,
1074
+ "eval_entropy": 0.6126393578695447,
1075
+ "eval_loss": 0.6706712245941162,
1076
+ "eval_mean_token_accuracy": 0.8155577500884452,
1077
+ "eval_num_tokens": 2368592.0,
1078
+ "eval_runtime": 88.1407,
1079
+ "eval_samples_per_second": 16.111,
1080
+ "eval_steps_per_second": 2.019,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.6072248566895724,
1085
+ "epoch": 2.506940253470127,
1086
+ "grad_norm": 0.6999370455741882,
1087
+ "learning_rate": 0.0003509614855443561,
1088
+ "loss": 0.5500593185424805,
1089
+ "mean_token_accuracy": 0.8337264291942119,
1090
+ "num_tokens": 2412859.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.506940253470127,
1095
+ "eval_entropy": 0.6417417074187418,
1096
+ "eval_loss": 0.6651163697242737,
1097
+ "eval_mean_token_accuracy": 0.8159113908081912,
1098
+ "eval_num_tokens": 2412859.0,
1099
+ "eval_runtime": 87.4854,
1100
+ "eval_samples_per_second": 16.231,
1101
+ "eval_steps_per_second": 2.035,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.6301592070609331,
1106
+ "epoch": 2.5552202776101387,
1107
+ "grad_norm": 0.8768910765647888,
1108
+ "learning_rate": 0.00034935254927581064,
1109
+ "loss": 0.5613903999328613,
1110
+ "mean_token_accuracy": 0.8298896946012974,
1111
+ "num_tokens": 2458201.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.5552202776101387,
1116
+ "eval_entropy": 0.6569794751285167,
1117
+ "eval_loss": 0.6668341159820557,
1118
+ "eval_mean_token_accuracy": 0.8126791795987761,
1119
+ "eval_num_tokens": 2458201.0,
1120
+ "eval_runtime": 88.0474,
1121
+ "eval_samples_per_second": 16.128,
1122
+ "eval_steps_per_second": 2.022,
1123
+ "step": 1060
1124
+ },
1125
+ {
1126
+ "entropy": 0.6069021724164486,
1127
+ "epoch": 2.603500301750151,
1128
+ "grad_norm": 0.9033508896827698,
1129
+ "learning_rate": 0.0003476979927158357,
1130
+ "loss": 0.5590654373168945,
1131
+ "mean_token_accuracy": 0.831520090252161,
1132
+ "num_tokens": 2505382.0,
1133
+ "step": 1080
1134
+ },
1135
+ {
1136
+ "epoch": 2.603500301750151,
1137
+ "eval_entropy": 0.6636380777600106,
1138
+ "eval_loss": 0.6596238017082214,
1139
+ "eval_mean_token_accuracy": 0.8177202826135614,
1140
+ "eval_num_tokens": 2505382.0,
1141
+ "eval_runtime": 88.08,
1142
+ "eval_samples_per_second": 16.122,
1143
+ "eval_steps_per_second": 2.021,
1144
+ "step": 1080
1145
+ },
1146
+ {
1147
+ "entropy": 0.6138021353632211,
1148
+ "epoch": 2.651780325890163,
1149
+ "grad_norm": 0.7345808148384094,
1150
+ "learning_rate": 0.0003459982840843664,
1151
+ "loss": 0.5586410522460937,
1152
+ "mean_token_accuracy": 0.8295876093208789,
1153
+ "num_tokens": 2552927.0,
1154
+ "step": 1100
1155
+ },
1156
+ {
1157
+ "epoch": 2.651780325890163,
1158
+ "eval_entropy": 0.5845904127600487,
1159
+ "eval_loss": 0.663780689239502,
1160
+ "eval_mean_token_accuracy": 0.8142095107710763,
1161
+ "eval_num_tokens": 2552927.0,
1162
+ "eval_runtime": 88.0083,
1163
+ "eval_samples_per_second": 16.135,
1164
+ "eval_steps_per_second": 2.023,
1165
+ "step": 1100
1166
+ },
1167
+ {
1168
+ "entropy": 0.6251844003796577,
1169
+ "epoch": 2.700060350030175,
1170
+ "grad_norm": 0.7587347626686096,
1171
+ "learning_rate": 0.00034425390437883976,
1172
+ "loss": 0.5720802307128906,
1173
+ "mean_token_accuracy": 0.8285771444439888,
1174
+ "num_tokens": 2597426.0,
1175
+ "step": 1120
1176
+ },
1177
+ {
1178
+ "epoch": 2.700060350030175,
1179
+ "eval_entropy": 0.6251207221759839,
1180
+ "eval_loss": 0.6556326746940613,
1181
+ "eval_mean_token_accuracy": 0.8181774036937886,
1182
+ "eval_num_tokens": 2597426.0,
1183
+ "eval_runtime": 87.8517,
1184
+ "eval_samples_per_second": 16.164,
1185
+ "eval_steps_per_second": 2.026,
1186
+ "step": 1120
1187
+ }
1188
+ ],
1189
+ "logging_steps": 20,
1190
+ "max_steps": 4150,
1191
+ "num_input_tokens_seen": 0,
1192
+ "num_train_epochs": 10,
1193
+ "save_steps": 20,
1194
+ "stateful_callbacks": {
1195
+ "TrainerControl": {
1196
+ "args": {
1197
+ "should_epoch_stop": false,
1198
+ "should_evaluate": false,
1199
+ "should_log": false,
1200
+ "should_save": true,
1201
+ "should_training_stop": false
1202
+ },
1203
+ "attributes": {}
1204
+ }
1205
+ },
1206
+ "total_flos": 1.0975532670678835e+17,
1207
+ "train_batch_size": 4,
1208
+ "trial_name": null,
1209
+ "trial_params": null
1210
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1140/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1140/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.03828026524568501,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "v_proj",
34
+ "up_proj",
35
+ "down_proj",
36
+ "o_proj",
37
+ "q_proj",
38
+ "k_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1140/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1140/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1140/trainer_state.json ADDED
@@ -0,0 +1,1231 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.748340374170187,
6
+ "eval_steps": 20,
7
+ "global_step": 1140,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 2.172999547421932,
14
+ "epoch": 0.04828002414001207,
15
+ "grad_norm": 2.5071208477020264,
16
+ "learning_rate": 1.7227585487708594e-05,
17
+ "loss": 1.9347333908081055,
18
+ "mean_token_accuracy": 0.6086816020309925,
19
+ "num_tokens": 47778.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.04828002414001207,
24
+ "eval_entropy": 1.6909754791956269,
25
+ "eval_loss": 1.4728385210037231,
26
+ "eval_mean_token_accuracy": 0.6773593797442619,
27
+ "eval_num_tokens": 47778.0,
28
+ "eval_runtime": 90.1265,
29
+ "eval_samples_per_second": 15.756,
30
+ "eval_steps_per_second": 1.975,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.2014626950025558,
35
+ "epoch": 0.09656004828002414,
36
+ "grad_norm": 1.5074543952941895,
37
+ "learning_rate": 3.5361886001086065e-05,
38
+ "loss": 1.0867146492004394,
39
+ "mean_token_accuracy": 0.7285927519202232,
40
+ "num_tokens": 95981.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.09656004828002414,
45
+ "eval_entropy": 1.0304168352250302,
46
+ "eval_loss": 0.9211422204971313,
47
+ "eval_mean_token_accuracy": 0.7613122848312507,
48
+ "eval_num_tokens": 95981.0,
49
+ "eval_runtime": 88.6338,
50
+ "eval_samples_per_second": 16.021,
51
+ "eval_steps_per_second": 2.008,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.9454198732972146,
56
+ "epoch": 0.14484007242003621,
57
+ "grad_norm": 1.7578108310699463,
58
+ "learning_rate": 5.349618651446354e-05,
59
+ "loss": 0.8447297096252442,
60
+ "mean_token_accuracy": 0.7723424732685089,
61
+ "num_tokens": 142784.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.14484007242003621,
66
+ "eval_entropy": 0.8647834474451086,
67
+ "eval_loss": 0.8240737318992615,
68
+ "eval_mean_token_accuracy": 0.7821227014064789,
69
+ "eval_num_tokens": 142784.0,
70
+ "eval_runtime": 88.0135,
71
+ "eval_samples_per_second": 16.134,
72
+ "eval_steps_per_second": 2.022,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.8963550612330436,
77
+ "epoch": 0.19312009656004828,
78
+ "grad_norm": 1.1464931964874268,
79
+ "learning_rate": 7.1630487027841e-05,
80
+ "loss": 0.7979836463928223,
81
+ "mean_token_accuracy": 0.7822489373385906,
82
+ "num_tokens": 187338.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.19312009656004828,
87
+ "eval_entropy": 0.8663296133614657,
88
+ "eval_loss": 0.7752988338470459,
89
+ "eval_mean_token_accuracy": 0.7885957012015782,
90
+ "eval_num_tokens": 187338.0,
91
+ "eval_runtime": 88.284,
92
+ "eval_samples_per_second": 16.084,
93
+ "eval_steps_per_second": 2.016,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.8265103206038475,
98
+ "epoch": 0.24140012070006034,
99
+ "grad_norm": 1.0106662511825562,
100
+ "learning_rate": 8.976478754121848e-05,
101
+ "loss": 0.7297060012817382,
102
+ "mean_token_accuracy": 0.7955138415098191,
103
+ "num_tokens": 236038.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24140012070006034,
108
+ "eval_entropy": 0.8122169536151244,
109
+ "eval_loss": 0.754906177520752,
110
+ "eval_mean_token_accuracy": 0.7917155255092664,
111
+ "eval_num_tokens": 236038.0,
112
+ "eval_runtime": 87.5754,
113
+ "eval_samples_per_second": 16.215,
114
+ "eval_steps_per_second": 2.033,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.8098392844200134,
119
+ "epoch": 0.28968014484007243,
120
+ "grad_norm": 1.2238041162490845,
121
+ "learning_rate": 0.00010789908805459595,
122
+ "loss": 0.7201489448547364,
123
+ "mean_token_accuracy": 0.7975016921758652,
124
+ "num_tokens": 282918.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.28968014484007243,
129
+ "eval_entropy": 0.8418726462326692,
130
+ "eval_loss": 0.7393178939819336,
131
+ "eval_mean_token_accuracy": 0.7936771585700217,
132
+ "eval_num_tokens": 282918.0,
133
+ "eval_runtime": 87.1597,
134
+ "eval_samples_per_second": 16.292,
135
+ "eval_steps_per_second": 2.042,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.8080591082572937,
140
+ "epoch": 0.33796016898008446,
141
+ "grad_norm": 1.011591911315918,
142
+ "learning_rate": 0.00012603338856797343,
143
+ "loss": 0.7147697925567627,
144
+ "mean_token_accuracy": 0.7947175532579422,
145
+ "num_tokens": 331305.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.33796016898008446,
150
+ "eval_entropy": 0.7894941431083037,
151
+ "eval_loss": 0.720219075679779,
152
+ "eval_mean_token_accuracy": 0.801899236239744,
153
+ "eval_num_tokens": 331305.0,
154
+ "eval_runtime": 87.1493,
155
+ "eval_samples_per_second": 16.294,
156
+ "eval_steps_per_second": 2.042,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.7957205109298229,
161
+ "epoch": 0.38624019312009655,
162
+ "grad_norm": 0.8481830954551697,
163
+ "learning_rate": 0.00014416768908135088,
164
+ "loss": 0.7050027847290039,
165
+ "mean_token_accuracy": 0.7995368830859662,
166
+ "num_tokens": 375876.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.38624019312009655,
171
+ "eval_entropy": 0.7746140591883928,
172
+ "eval_loss": 0.7175475358963013,
173
+ "eval_mean_token_accuracy": 0.7994848955213354,
174
+ "eval_num_tokens": 375876.0,
175
+ "eval_runtime": 86.655,
176
+ "eval_samples_per_second": 16.387,
177
+ "eval_steps_per_second": 2.054,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.779565304517746,
182
+ "epoch": 0.43452021726010864,
183
+ "grad_norm": 1.032624363899231,
184
+ "learning_rate": 0.00016230198959472835,
185
+ "loss": 0.6978332042694092,
186
+ "mean_token_accuracy": 0.8015618294477462,
187
+ "num_tokens": 423913.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.43452021726010864,
192
+ "eval_entropy": 0.7455583871080634,
193
+ "eval_loss": 0.7118850350379944,
194
+ "eval_mean_token_accuracy": 0.7999682195400923,
195
+ "eval_num_tokens": 423913.0,
196
+ "eval_runtime": 86.7842,
197
+ "eval_samples_per_second": 16.362,
198
+ "eval_steps_per_second": 2.051,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.7810172818601131,
203
+ "epoch": 0.4828002414001207,
204
+ "grad_norm": 0.9276663064956665,
205
+ "learning_rate": 0.00018043629010810582,
206
+ "loss": 0.6937861442565918,
207
+ "mean_token_accuracy": 0.802893053740263,
208
+ "num_tokens": 472395.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.4828002414001207,
213
+ "eval_entropy": 0.7650193290764027,
214
+ "eval_loss": 0.7020567059516907,
215
+ "eval_mean_token_accuracy": 0.8045302153973097,
216
+ "eval_num_tokens": 472395.0,
217
+ "eval_runtime": 88.326,
218
+ "eval_samples_per_second": 16.077,
219
+ "eval_steps_per_second": 2.015,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.7862060204148292,
224
+ "epoch": 0.5310802655401328,
225
+ "grad_norm": 0.8195295929908752,
226
+ "learning_rate": 0.0001985705906214833,
227
+ "loss": 0.6878782272338867,
228
+ "mean_token_accuracy": 0.8027168907225132,
229
+ "num_tokens": 517049.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.5310802655401328,
234
+ "eval_entropy": 0.7656565583154057,
235
+ "eval_loss": 0.6973932981491089,
236
+ "eval_mean_token_accuracy": 0.8047967278555538,
237
+ "eval_num_tokens": 517049.0,
238
+ "eval_runtime": 87.7698,
239
+ "eval_samples_per_second": 16.179,
240
+ "eval_steps_per_second": 2.028,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.7611534461379051,
245
+ "epoch": 0.5793602896801449,
246
+ "grad_norm": 0.7499712109565735,
247
+ "learning_rate": 0.00021670489113486077,
248
+ "loss": 0.681937837600708,
249
+ "mean_token_accuracy": 0.8041978515684605,
250
+ "num_tokens": 562836.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.5793602896801449,
255
+ "eval_entropy": 0.7599442380197933,
256
+ "eval_loss": 0.7002046704292297,
257
+ "eval_mean_token_accuracy": 0.8051861517884759,
258
+ "eval_num_tokens": 562836.0,
259
+ "eval_runtime": 87.9239,
260
+ "eval_samples_per_second": 16.15,
261
+ "eval_steps_per_second": 2.024,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.7800783462822437,
266
+ "epoch": 0.627640313820157,
267
+ "grad_norm": 0.8645269274711609,
268
+ "learning_rate": 0.00023483919164823825,
269
+ "loss": 0.6954500675201416,
270
+ "mean_token_accuracy": 0.8031484372913837,
271
+ "num_tokens": 606612.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.627640313820157,
276
+ "eval_entropy": 0.7498679090751691,
277
+ "eval_loss": 0.7046149373054504,
278
+ "eval_mean_token_accuracy": 0.802388215667746,
279
+ "eval_num_tokens": 606612.0,
280
+ "eval_runtime": 88.2468,
281
+ "eval_samples_per_second": 16.091,
282
+ "eval_steps_per_second": 2.017,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.7586533665657044,
287
+ "epoch": 0.6759203379601689,
288
+ "grad_norm": 0.741596519947052,
289
+ "learning_rate": 0.0002529734921616157,
290
+ "loss": 0.690476369857788,
291
+ "mean_token_accuracy": 0.8046923100948333,
292
+ "num_tokens": 655920.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6759203379601689,
297
+ "eval_entropy": 0.7714274066887544,
298
+ "eval_loss": 0.6992160677909851,
299
+ "eval_mean_token_accuracy": 0.8042646482419432,
300
+ "eval_num_tokens": 655920.0,
301
+ "eval_runtime": 88.1754,
302
+ "eval_samples_per_second": 16.104,
303
+ "eval_steps_per_second": 2.019,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.7585848286747933,
308
+ "epoch": 0.724200362100181,
309
+ "grad_norm": 1.0778124332427979,
310
+ "learning_rate": 0.00027110779267499317,
311
+ "loss": 0.6790409564971924,
312
+ "mean_token_accuracy": 0.8060351841151714,
313
+ "num_tokens": 702638.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.724200362100181,
318
+ "eval_entropy": 0.7380402811457601,
319
+ "eval_loss": 0.7054820656776428,
320
+ "eval_mean_token_accuracy": 0.8003534738267406,
321
+ "eval_num_tokens": 702638.0,
322
+ "eval_runtime": 88.238,
323
+ "eval_samples_per_second": 16.093,
324
+ "eval_steps_per_second": 2.017,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.7696981988847256,
329
+ "epoch": 0.7724803862401931,
330
+ "grad_norm": 0.8408658504486084,
331
+ "learning_rate": 0.00028924209318837064,
332
+ "loss": 0.7070387363433838,
333
+ "mean_token_accuracy": 0.8035947173833847,
334
+ "num_tokens": 749377.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.7724803862401931,
339
+ "eval_entropy": 0.7897409023193831,
340
+ "eval_loss": 0.7015668749809265,
341
+ "eval_mean_token_accuracy": 0.805306687113944,
342
+ "eval_num_tokens": 749377.0,
343
+ "eval_runtime": 87.9395,
344
+ "eval_samples_per_second": 16.147,
345
+ "eval_steps_per_second": 2.024,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.762337189912796,
350
+ "epoch": 0.8207604103802052,
351
+ "grad_norm": 0.9843310117721558,
352
+ "learning_rate": 0.0003073763937017481,
353
+ "loss": 0.7021088123321533,
354
+ "mean_token_accuracy": 0.8029616877436638,
355
+ "num_tokens": 798874.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8207604103802052,
360
+ "eval_entropy": 0.7210077121016685,
361
+ "eval_loss": 0.7082746624946594,
362
+ "eval_mean_token_accuracy": 0.8049283824609906,
363
+ "eval_num_tokens": 798874.0,
364
+ "eval_runtime": 88.1706,
365
+ "eval_samples_per_second": 16.105,
366
+ "eval_steps_per_second": 2.019,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.7879025422036647,
371
+ "epoch": 0.8690404345202173,
372
+ "grad_norm": 0.9713814854621887,
373
+ "learning_rate": 0.0003255106942151256,
374
+ "loss": 0.714624547958374,
375
+ "mean_token_accuracy": 0.79987031519413,
376
+ "num_tokens": 840825.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8690404345202173,
381
+ "eval_entropy": 0.7866930489459735,
382
+ "eval_loss": 0.7145519852638245,
383
+ "eval_mean_token_accuracy": 0.8015657988157165,
384
+ "eval_num_tokens": 840825.0,
385
+ "eval_runtime": 88.3746,
386
+ "eval_samples_per_second": 16.068,
387
+ "eval_steps_per_second": 2.014,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.7844726584851742,
392
+ "epoch": 0.9173204586602294,
393
+ "grad_norm": 1.2028311491012573,
394
+ "learning_rate": 0.00034364499472850306,
395
+ "loss": 0.7040786266326904,
396
+ "mean_token_accuracy": 0.7981885217130185,
397
+ "num_tokens": 882885.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9173204586602294,
402
+ "eval_entropy": 0.7767668339643585,
403
+ "eval_loss": 0.7073574066162109,
404
+ "eval_mean_token_accuracy": 0.8040851833445303,
405
+ "eval_num_tokens": 882885.0,
406
+ "eval_runtime": 88.3828,
407
+ "eval_samples_per_second": 16.066,
408
+ "eval_steps_per_second": 2.014,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.7699790760874748,
413
+ "epoch": 0.9656004828002414,
414
+ "grad_norm": 0.9397398233413696,
415
+ "learning_rate": 0.00036177929524188054,
416
+ "loss": 0.7040313720703125,
417
+ "mean_token_accuracy": 0.8034727744758129,
418
+ "num_tokens": 927456.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9656004828002414,
423
+ "eval_entropy": 0.7749103545472863,
424
+ "eval_loss": 0.7093513011932373,
425
+ "eval_mean_token_accuracy": 0.8019482901926791,
426
+ "eval_num_tokens": 927456.0,
427
+ "eval_runtime": 86.2112,
428
+ "eval_samples_per_second": 16.471,
429
+ "eval_steps_per_second": 2.065,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.7672389172888422,
434
+ "epoch": 1.012070006035003,
435
+ "grad_norm": 0.99210524559021,
436
+ "learning_rate": 0.00037628567078152296,
437
+ "loss": 0.6913118362426758,
438
+ "mean_token_accuracy": 0.8042210944287189,
439
+ "num_tokens": 972662.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.012070006035003,
444
+ "eval_entropy": 0.7294771229283193,
445
+ "eval_loss": 0.7171520590782166,
446
+ "eval_mean_token_accuracy": 0.8015128585059991,
447
+ "eval_num_tokens": 972662.0,
448
+ "eval_runtime": 88.037,
449
+ "eval_samples_per_second": 16.13,
450
+ "eval_steps_per_second": 2.022,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.7143943183124065,
455
+ "epoch": 1.060350030175015,
456
+ "grad_norm": 1.087638258934021,
457
+ "learning_rate": 0.0003762484015601069,
458
+ "loss": 0.642704439163208,
459
+ "mean_token_accuracy": 0.8124966286122799,
460
+ "num_tokens": 1022525.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.060350030175015,
465
+ "eval_entropy": 0.743511886744017,
466
+ "eval_loss": 0.7116662859916687,
467
+ "eval_mean_token_accuracy": 0.8037210672758939,
468
+ "eval_num_tokens": 1022525.0,
469
+ "eval_runtime": 87.9931,
470
+ "eval_samples_per_second": 16.138,
471
+ "eval_steps_per_second": 2.023,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.7237346574664116,
476
+ "epoch": 1.1086300543150271,
477
+ "grad_norm": 0.9365553259849548,
478
+ "learning_rate": 0.0003761579008392123,
479
+ "loss": 0.652644681930542,
480
+ "mean_token_accuracy": 0.8107496216893196,
481
+ "num_tokens": 1069194.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1086300543150271,
486
+ "eval_entropy": 0.7372823839776972,
487
+ "eval_loss": 0.7238303422927856,
488
+ "eval_mean_token_accuracy": 0.8026230124275336,
489
+ "eval_num_tokens": 1069194.0,
490
+ "eval_runtime": 87.6778,
491
+ "eval_samples_per_second": 16.196,
492
+ "eval_steps_per_second": 2.03,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.7202137842774391,
497
+ "epoch": 1.1569100784550392,
498
+ "grad_norm": 0.7998443841934204,
499
+ "learning_rate": 0.0003760141942294726,
500
+ "loss": 0.6485219478607178,
501
+ "mean_token_accuracy": 0.8092714451253414,
502
+ "num_tokens": 1117570.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1569100784550392,
507
+ "eval_entropy": 0.71547647942318,
508
+ "eval_loss": 0.7124277949333191,
509
+ "eval_mean_token_accuracy": 0.8013174219077892,
510
+ "eval_num_tokens": 1117570.0,
511
+ "eval_runtime": 87.4387,
512
+ "eval_samples_per_second": 16.24,
513
+ "eval_steps_per_second": 2.036,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.7270819500088692,
518
+ "epoch": 1.2051901025950513,
519
+ "grad_norm": 0.9470787644386292,
520
+ "learning_rate": 0.00037581732239815854,
521
+ "loss": 0.66352219581604,
522
+ "mean_token_accuracy": 0.8102593503892421,
523
+ "num_tokens": 1161717.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.2051901025950513,
528
+ "eval_entropy": 0.7791565897759427,
529
+ "eval_loss": 0.7179726958274841,
530
+ "eval_mean_token_accuracy": 0.7948898328154275,
531
+ "eval_num_tokens": 1161717.0,
532
+ "eval_runtime": 87.9927,
533
+ "eval_samples_per_second": 16.138,
534
+ "eval_steps_per_second": 2.023,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.7320936664938926,
539
+ "epoch": 1.2534701267350634,
540
+ "grad_norm": 1.0220379829406738,
541
+ "learning_rate": 0.0003755673410576695,
542
+ "loss": 0.6581549644470215,
543
+ "mean_token_accuracy": 0.8098709337413311,
544
+ "num_tokens": 1209390.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.2534701267350634,
549
+ "eval_entropy": 0.7710678225822662,
550
+ "eval_loss": 0.708797812461853,
551
+ "eval_mean_token_accuracy": 0.8039572778042783,
552
+ "eval_num_tokens": 1209390.0,
553
+ "eval_runtime": 88.1193,
554
+ "eval_samples_per_second": 16.115,
555
+ "eval_steps_per_second": 2.02,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.7281674664467573,
560
+ "epoch": 1.3017501508750755,
561
+ "grad_norm": 0.8775188326835632,
562
+ "learning_rate": 0.000375264320949768,
563
+ "loss": 0.6631278991699219,
564
+ "mean_token_accuracy": 0.8101261422038079,
565
+ "num_tokens": 1256611.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3017501508750755,
570
+ "eval_entropy": 0.7199799800187014,
571
+ "eval_loss": 0.6981531381607056,
572
+ "eval_mean_token_accuracy": 0.8071265254127845,
573
+ "eval_num_tokens": 1256611.0,
574
+ "eval_runtime": 88.165,
575
+ "eval_samples_per_second": 16.106,
576
+ "eval_steps_per_second": 2.019,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.722860773652792,
581
+ "epoch": 1.3500301750150876,
582
+ "grad_norm": 0.8949288129806519,
583
+ "learning_rate": 0.00037490834782556,
584
+ "loss": 0.6639598846435547,
585
+ "mean_token_accuracy": 0.8105649061501026,
586
+ "num_tokens": 1302325.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3500301750150876,
591
+ "eval_entropy": 0.7134760032209118,
592
+ "eval_loss": 0.7029948234558105,
593
+ "eval_mean_token_accuracy": 0.8062985880991046,
594
+ "eval_num_tokens": 1302325.0,
595
+ "eval_runtime": 88.1286,
596
+ "eval_samples_per_second": 16.113,
597
+ "eval_steps_per_second": 2.02,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.7299126699566841,
602
+ "epoch": 1.3983101991550995,
603
+ "grad_norm": 1.0323039293289185,
604
+ "learning_rate": 0.0003744995224212291,
605
+ "loss": 0.6535763263702392,
606
+ "mean_token_accuracy": 0.8120224848389626,
607
+ "num_tokens": 1343844.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.3983101991550995,
612
+ "eval_entropy": 0.7169701182440426,
613
+ "eval_loss": 0.7067857980728149,
614
+ "eval_mean_token_accuracy": 0.8040432936689826,
615
+ "eval_num_tokens": 1343844.0,
616
+ "eval_runtime": 88.2252,
617
+ "eval_samples_per_second": 16.095,
618
+ "eval_steps_per_second": 2.018,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.7281742256134749,
623
+ "epoch": 1.4465902232951118,
624
+ "grad_norm": 1.3370424509048462,
625
+ "learning_rate": 0.00037403796042952863,
626
+ "loss": 0.6637272357940673,
627
+ "mean_token_accuracy": 0.8118261776864528,
628
+ "num_tokens": 1389471.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4465902232951118,
633
+ "eval_entropy": 0.7378843212395572,
634
+ "eval_loss": 0.7038390040397644,
635
+ "eval_mean_token_accuracy": 0.8059229371922739,
636
+ "eval_num_tokens": 1389471.0,
637
+ "eval_runtime": 87.9301,
638
+ "eval_samples_per_second": 16.149,
639
+ "eval_steps_per_second": 2.024,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.7388614989817143,
644
+ "epoch": 1.4948702474351236,
645
+ "grad_norm": 0.8464171290397644,
646
+ "learning_rate": 0.00037352379246704257,
647
+ "loss": 0.6604740142822265,
648
+ "mean_token_accuracy": 0.8102974124252796,
649
+ "num_tokens": 1435975.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.4948702474351236,
654
+ "eval_entropy": 0.7126847046814607,
655
+ "eval_loss": 0.6950703263282776,
656
+ "eval_mean_token_accuracy": 0.8085885392815879,
657
+ "eval_num_tokens": 1435975.0,
658
+ "eval_runtime": 88.2732,
659
+ "eval_samples_per_second": 16.086,
660
+ "eval_steps_per_second": 2.016,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.7258916139602661,
665
+ "epoch": 1.5431502715751357,
666
+ "grad_norm": 1.1780085563659668,
667
+ "learning_rate": 0.0003729571640372223,
668
+ "loss": 0.6529891014099121,
669
+ "mean_token_accuracy": 0.8085981778800487,
670
+ "num_tokens": 1482035.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.5431502715751357,
675
+ "eval_entropy": 0.7096509638797032,
676
+ "eval_loss": 0.690051257610321,
677
+ "eval_mean_token_accuracy": 0.8100520241796301,
678
+ "eval_num_tokens": 1482035.0,
679
+ "eval_runtime": 88.3153,
680
+ "eval_samples_per_second": 16.079,
681
+ "eval_steps_per_second": 2.016,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.7151173129677773,
686
+ "epoch": 1.5914302957151478,
687
+ "grad_norm": 1.0357667207717896,
688
+ "learning_rate": 0.0003723382354892108,
689
+ "loss": 0.6451474189758301,
690
+ "mean_token_accuracy": 0.8136902332305909,
691
+ "num_tokens": 1528119.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.5914302957151478,
696
+ "eval_entropy": 0.7060252652409371,
697
+ "eval_loss": 0.6899483799934387,
698
+ "eval_mean_token_accuracy": 0.810834194502134,
699
+ "eval_num_tokens": 1528119.0,
700
+ "eval_runtime": 87.8848,
701
+ "eval_samples_per_second": 16.158,
702
+ "eval_steps_per_second": 2.025,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.7141377851366997,
707
+ "epoch": 1.63971031985516,
708
+ "grad_norm": 0.8391667008399963,
709
+ "learning_rate": 0.000371667181972466,
710
+ "loss": 0.6481579780578614,
711
+ "mean_token_accuracy": 0.8159015104174614,
712
+ "num_tokens": 1573661.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.63971031985516,
717
+ "eval_entropy": 0.7598993319473909,
718
+ "eval_loss": 0.681204080581665,
719
+ "eval_mean_token_accuracy": 0.8100194586126992,
720
+ "eval_num_tokens": 1573661.0,
721
+ "eval_runtime": 88.2415,
722
+ "eval_samples_per_second": 16.092,
723
+ "eval_steps_per_second": 2.017,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.7235980287194252,
728
+ "epoch": 1.687990343995172,
729
+ "grad_norm": 0.8197196125984192,
730
+ "learning_rate": 0.00037094419338719537,
731
+ "loss": 0.6582849025726318,
732
+ "mean_token_accuracy": 0.8114834539592266,
733
+ "num_tokens": 1621714.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.687990343995172,
738
+ "eval_entropy": 0.7166854568411795,
739
+ "eval_loss": 0.679177463054657,
740
+ "eval_mean_token_accuracy": 0.8119039877077167,
741
+ "eval_num_tokens": 1621714.0,
742
+ "eval_runtime": 88.0957,
743
+ "eval_samples_per_second": 16.119,
744
+ "eval_steps_per_second": 2.021,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.7278190158307553,
749
+ "epoch": 1.736270368135184,
750
+ "grad_norm": 0.7820602655410767,
751
+ "learning_rate": 0.0003701694743306164,
752
+ "loss": 0.6633153438568116,
753
+ "mean_token_accuracy": 0.8123249113559723,
754
+ "num_tokens": 1672118.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.736270368135184,
759
+ "eval_entropy": 0.6980850314826108,
760
+ "eval_loss": 0.6775653958320618,
761
+ "eval_mean_token_accuracy": 0.8116872906684875,
762
+ "eval_num_tokens": 1672118.0,
763
+ "eval_runtime": 87.9795,
764
+ "eval_samples_per_second": 16.14,
765
+ "eval_steps_per_second": 2.023,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.7149631556123495,
770
+ "epoch": 1.7845503922751962,
771
+ "grad_norm": 1.0139760971069336,
772
+ "learning_rate": 0.00036934324403905795,
773
+ "loss": 0.6468976020812989,
774
+ "mean_token_accuracy": 0.8117561548948288,
775
+ "num_tokens": 1720465.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.7845503922751962,
780
+ "eval_entropy": 0.69009849094273,
781
+ "eval_loss": 0.6731093525886536,
782
+ "eval_mean_token_accuracy": 0.81281964196248,
783
+ "eval_num_tokens": 1720465.0,
784
+ "eval_runtime": 87.7412,
785
+ "eval_samples_per_second": 16.184,
786
+ "eval_steps_per_second": 2.029,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.7226049453020096,
791
+ "epoch": 1.832830416415208,
792
+ "grad_norm": 1.5040546655654907,
793
+ "learning_rate": 0.0003684657363259193,
794
+ "loss": 0.6385328769683838,
795
+ "mean_token_accuracy": 0.8132639616727829,
796
+ "num_tokens": 1764440.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.832830416415208,
801
+ "eval_entropy": 0.7125169618076153,
802
+ "eval_loss": 0.6864680647850037,
803
+ "eval_mean_token_accuracy": 0.8049418089095126,
804
+ "eval_num_tokens": 1764440.0,
805
+ "eval_runtime": 87.9429,
806
+ "eval_samples_per_second": 16.147,
807
+ "eval_steps_per_second": 2.024,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.707821810990572,
812
+ "epoch": 1.8811104405552204,
813
+ "grad_norm": 0.8604223132133484,
814
+ "learning_rate": 0.00036753719951550327,
815
+ "loss": 0.6397049427032471,
816
+ "mean_token_accuracy": 0.8137877315282822,
817
+ "num_tokens": 1810828.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.8811104405552204,
822
+ "eval_entropy": 0.6492102551326323,
823
+ "eval_loss": 0.6743726134300232,
824
+ "eval_mean_token_accuracy": 0.8130356572317273,
825
+ "eval_num_tokens": 1810828.0,
826
+ "eval_runtime": 87.8684,
827
+ "eval_samples_per_second": 16.161,
828
+ "eval_steps_per_second": 2.026,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.6800355084240437,
833
+ "epoch": 1.9293904646952322,
834
+ "grad_norm": 0.7270982265472412,
835
+ "learning_rate": 0.00036655789637274377,
836
+ "loss": 0.6267755031585693,
837
+ "mean_token_accuracy": 0.8180945307016373,
838
+ "num_tokens": 1857600.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9293904646952322,
843
+ "eval_entropy": 0.6727538651294922,
844
+ "eval_loss": 0.6696028709411621,
845
+ "eval_mean_token_accuracy": 0.8116443090224534,
846
+ "eval_num_tokens": 1857600.0,
847
+ "eval_runtime": 88.2449,
848
+ "eval_samples_per_second": 16.092,
849
+ "eval_steps_per_second": 2.017,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.6746691606938839,
854
+ "epoch": 1.9776704888352445,
855
+ "grad_norm": 0.9771540760993958,
856
+ "learning_rate": 0.0003655281040288461,
857
+ "loss": 0.6326992988586426,
858
+ "mean_token_accuracy": 0.8179976396262646,
859
+ "num_tokens": 1901998.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 1.9776704888352445,
864
+ "eval_entropy": 0.6636319716324967,
865
+ "eval_loss": 0.6714363098144531,
866
+ "eval_mean_token_accuracy": 0.8126635943235976,
867
+ "eval_num_tokens": 1901998.0,
868
+ "eval_runtime": 87.7666,
869
+ "eval_samples_per_second": 16.179,
870
+ "eval_steps_per_second": 2.028,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.6319125778489298,
875
+ "epoch": 2.024140012070006,
876
+ "grad_norm": 0.8380444645881653,
877
+ "learning_rate": 0.0003644481139028622,
878
+ "loss": 0.5780903816223144,
879
+ "mean_token_accuracy": 0.8250188401767186,
880
+ "num_tokens": 1947553.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.024140012070006,
885
+ "eval_entropy": 0.6627151989535,
886
+ "eval_loss": 0.6774935722351074,
887
+ "eval_mean_token_accuracy": 0.8140510819601209,
888
+ "eval_num_tokens": 1947553.0,
889
+ "eval_runtime": 87.8287,
890
+ "eval_samples_per_second": 16.168,
891
+ "eval_steps_per_second": 2.027,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.6059147048741579,
896
+ "epoch": 2.0724200362100182,
897
+ "grad_norm": 0.9745152592658997,
898
+ "learning_rate": 0.0003633182316192229,
899
+ "loss": 0.546702527999878,
900
+ "mean_token_accuracy": 0.8358186542987823,
901
+ "num_tokens": 1995234.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.0724200362100182,
906
+ "eval_entropy": 0.6620710384979677,
907
+ "eval_loss": 0.6769291162490845,
908
+ "eval_mean_token_accuracy": 0.8135082835561773,
909
+ "eval_num_tokens": 1995234.0,
910
+ "eval_runtime": 88.0008,
911
+ "eval_samples_per_second": 16.136,
912
+ "eval_steps_per_second": 2.023,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.6076117843389511,
917
+ "epoch": 2.12070006035003,
918
+ "grad_norm": 0.6914380192756653,
919
+ "learning_rate": 0.0003621387769212491,
920
+ "loss": 0.5324535846710206,
921
+ "mean_token_accuracy": 0.8366568259894848,
922
+ "num_tokens": 2040639.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.12070006035003,
927
+ "eval_entropy": 0.6592224213514435,
928
+ "eval_loss": 0.670383632183075,
929
+ "eval_mean_token_accuracy": 0.8162150088320957,
930
+ "eval_num_tokens": 2040639.0,
931
+ "eval_runtime": 88.324,
932
+ "eval_samples_per_second": 16.077,
933
+ "eval_steps_per_second": 2.015,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.6158072650432587,
938
+ "epoch": 2.1689800844900424,
939
+ "grad_norm": 0.9649198055267334,
940
+ "learning_rate": 0.0003609100835806688,
941
+ "loss": 0.5564568996429443,
942
+ "mean_token_accuracy": 0.8329654954373836,
943
+ "num_tokens": 2090604.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.1689800844900424,
948
+ "eval_entropy": 0.608588819758276,
949
+ "eval_loss": 0.6745102405548096,
950
+ "eval_mean_token_accuracy": 0.8155080542135774,
951
+ "eval_num_tokens": 2090604.0,
952
+ "eval_runtime": 87.52,
953
+ "eval_samples_per_second": 16.225,
954
+ "eval_steps_per_second": 2.034,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.5938734326511621,
959
+ "epoch": 2.2172601086300543,
960
+ "grad_norm": 0.6915557384490967,
961
+ "learning_rate": 0.0003596324993031632,
962
+ "loss": 0.5351875782012939,
963
+ "mean_token_accuracy": 0.8385631121695042,
964
+ "num_tokens": 2142246.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2172601086300543,
969
+ "eval_entropy": 0.6251631205001574,
970
+ "eval_loss": 0.6664518117904663,
971
+ "eval_mean_token_accuracy": 0.81711940417129,
972
+ "eval_num_tokens": 2142246.0,
973
+ "eval_runtime": 87.1792,
974
+ "eval_samples_per_second": 16.288,
975
+ "eval_steps_per_second": 2.042,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.623173613846302,
980
+ "epoch": 2.2655401327700666,
981
+ "grad_norm": 0.8366610407829285,
982
+ "learning_rate": 0.00035830638562997063,
983
+ "loss": 0.5488213539123535,
984
+ "mean_token_accuracy": 0.8321483485400677,
985
+ "num_tokens": 2183144.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.2655401327700666,
990
+ "eval_entropy": 0.6237085649136747,
991
+ "eval_loss": 0.68075031042099,
992
+ "eval_mean_token_accuracy": 0.814671738429016,
993
+ "eval_num_tokens": 2183144.0,
994
+ "eval_runtime": 88.0652,
995
+ "eval_samples_per_second": 16.124,
996
+ "eval_steps_per_second": 2.021,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.6236911740154027,
1001
+ "epoch": 2.3138201569100785,
1002
+ "grad_norm": 0.7812452912330627,
1003
+ "learning_rate": 0.00035693211783557416,
1004
+ "loss": 0.5696200847625732,
1005
+ "mean_token_accuracy": 0.82991396561265,
1006
+ "num_tokens": 2228098.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3138201569100785,
1011
+ "eval_entropy": 0.6327033426319615,
1012
+ "eval_loss": 0.6742984652519226,
1013
+ "eval_mean_token_accuracy": 0.8148086091105857,
1014
+ "eval_num_tokens": 2228098.0,
1015
+ "eval_runtime": 88.1711,
1016
+ "eval_samples_per_second": 16.105,
1017
+ "eval_steps_per_second": 2.019,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.6218426413834095,
1022
+ "epoch": 2.3621001810500903,
1023
+ "grad_norm": 0.8719790577888489,
1024
+ "learning_rate": 0.0003555100848215035,
1025
+ "loss": 0.5582651615142822,
1026
+ "mean_token_accuracy": 0.8323961742222309,
1027
+ "num_tokens": 2273062.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.3621001810500903,
1032
+ "eval_entropy": 0.6338012375858393,
1033
+ "eval_loss": 0.6685822606086731,
1034
+ "eval_mean_token_accuracy": 0.8152180300669724,
1035
+ "eval_num_tokens": 2273062.0,
1036
+ "eval_runtime": 88.101,
1037
+ "eval_samples_per_second": 16.118,
1038
+ "eval_steps_per_second": 2.02,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.6050681818276644,
1043
+ "epoch": 2.4103802051901027,
1044
+ "grad_norm": 0.9151566028594971,
1045
+ "learning_rate": 0.00035404068900628076,
1046
+ "loss": 0.559452486038208,
1047
+ "mean_token_accuracy": 0.8310242861509323,
1048
+ "num_tokens": 2319305.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.4103802051901027,
1053
+ "eval_entropy": 0.6634890969549672,
1054
+ "eval_loss": 0.6692465543746948,
1055
+ "eval_mean_token_accuracy": 0.8134353673190213,
1056
+ "eval_num_tokens": 2319305.0,
1057
+ "eval_runtime": 87.7711,
1058
+ "eval_samples_per_second": 16.178,
1059
+ "eval_steps_per_second": 2.028,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.6169968567788601,
1064
+ "epoch": 2.4586602293301145,
1065
+ "grad_norm": 0.7081006765365601,
1066
+ "learning_rate": 0.0003525243462115402,
1067
+ "loss": 0.5589694023132324,
1068
+ "mean_token_accuracy": 0.8357502184808254,
1069
+ "num_tokens": 2368592.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.4586602293301145,
1074
+ "eval_entropy": 0.6126393578695447,
1075
+ "eval_loss": 0.6706712245941162,
1076
+ "eval_mean_token_accuracy": 0.8155577500884452,
1077
+ "eval_num_tokens": 2368592.0,
1078
+ "eval_runtime": 88.1407,
1079
+ "eval_samples_per_second": 16.111,
1080
+ "eval_steps_per_second": 2.019,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.6072248566895724,
1085
+ "epoch": 2.506940253470127,
1086
+ "grad_norm": 0.6999370455741882,
1087
+ "learning_rate": 0.0003509614855443561,
1088
+ "loss": 0.5500593185424805,
1089
+ "mean_token_accuracy": 0.8337264291942119,
1090
+ "num_tokens": 2412859.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.506940253470127,
1095
+ "eval_entropy": 0.6417417074187418,
1096
+ "eval_loss": 0.6651163697242737,
1097
+ "eval_mean_token_accuracy": 0.8159113908081912,
1098
+ "eval_num_tokens": 2412859.0,
1099
+ "eval_runtime": 87.4854,
1100
+ "eval_samples_per_second": 16.231,
1101
+ "eval_steps_per_second": 2.035,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.6301592070609331,
1106
+ "epoch": 2.5552202776101387,
1107
+ "grad_norm": 0.8768910765647888,
1108
+ "learning_rate": 0.00034935254927581064,
1109
+ "loss": 0.5613903999328613,
1110
+ "mean_token_accuracy": 0.8298896946012974,
1111
+ "num_tokens": 2458201.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.5552202776101387,
1116
+ "eval_entropy": 0.6569794751285167,
1117
+ "eval_loss": 0.6668341159820557,
1118
+ "eval_mean_token_accuracy": 0.8126791795987761,
1119
+ "eval_num_tokens": 2458201.0,
1120
+ "eval_runtime": 88.0474,
1121
+ "eval_samples_per_second": 16.128,
1122
+ "eval_steps_per_second": 2.022,
1123
+ "step": 1060
1124
+ },
1125
+ {
1126
+ "entropy": 0.6069021724164486,
1127
+ "epoch": 2.603500301750151,
1128
+ "grad_norm": 0.9033508896827698,
1129
+ "learning_rate": 0.0003476979927158357,
1130
+ "loss": 0.5590654373168945,
1131
+ "mean_token_accuracy": 0.831520090252161,
1132
+ "num_tokens": 2505382.0,
1133
+ "step": 1080
1134
+ },
1135
+ {
1136
+ "epoch": 2.603500301750151,
1137
+ "eval_entropy": 0.6636380777600106,
1138
+ "eval_loss": 0.6596238017082214,
1139
+ "eval_mean_token_accuracy": 0.8177202826135614,
1140
+ "eval_num_tokens": 2505382.0,
1141
+ "eval_runtime": 88.08,
1142
+ "eval_samples_per_second": 16.122,
1143
+ "eval_steps_per_second": 2.021,
1144
+ "step": 1080
1145
+ },
1146
+ {
1147
+ "entropy": 0.6138021353632211,
1148
+ "epoch": 2.651780325890163,
1149
+ "grad_norm": 0.7345808148384094,
1150
+ "learning_rate": 0.0003459982840843664,
1151
+ "loss": 0.5586410522460937,
1152
+ "mean_token_accuracy": 0.8295876093208789,
1153
+ "num_tokens": 2552927.0,
1154
+ "step": 1100
1155
+ },
1156
+ {
1157
+ "epoch": 2.651780325890163,
1158
+ "eval_entropy": 0.5845904127600487,
1159
+ "eval_loss": 0.663780689239502,
1160
+ "eval_mean_token_accuracy": 0.8142095107710763,
1161
+ "eval_num_tokens": 2552927.0,
1162
+ "eval_runtime": 88.0083,
1163
+ "eval_samples_per_second": 16.135,
1164
+ "eval_steps_per_second": 2.023,
1165
+ "step": 1100
1166
+ },
1167
+ {
1168
+ "entropy": 0.6251844003796577,
1169
+ "epoch": 2.700060350030175,
1170
+ "grad_norm": 0.7587347626686096,
1171
+ "learning_rate": 0.00034425390437883976,
1172
+ "loss": 0.5720802307128906,
1173
+ "mean_token_accuracy": 0.8285771444439888,
1174
+ "num_tokens": 2597426.0,
1175
+ "step": 1120
1176
+ },
1177
+ {
1178
+ "epoch": 2.700060350030175,
1179
+ "eval_entropy": 0.6251207221759839,
1180
+ "eval_loss": 0.6556326746940613,
1181
+ "eval_mean_token_accuracy": 0.8181774036937886,
1182
+ "eval_num_tokens": 2597426.0,
1183
+ "eval_runtime": 87.8517,
1184
+ "eval_samples_per_second": 16.164,
1185
+ "eval_steps_per_second": 2.026,
1186
+ "step": 1120
1187
+ },
1188
+ {
1189
+ "entropy": 0.6227757651358843,
1190
+ "epoch": 2.748340374170187,
1191
+ "grad_norm": 0.9254975914955139,
1192
+ "learning_rate": 0.00034246534723807843,
1193
+ "loss": 0.5767871856689453,
1194
+ "mean_token_accuracy": 0.8292960874736309,
1195
+ "num_tokens": 2640514.0,
1196
+ "step": 1140
1197
+ },
1198
+ {
1199
+ "epoch": 2.748340374170187,
1200
+ "eval_entropy": 0.6580858331048087,
1201
+ "eval_loss": 0.6579018831253052,
1202
+ "eval_mean_token_accuracy": 0.8178701273510965,
1203
+ "eval_num_tokens": 2640514.0,
1204
+ "eval_runtime": 87.9523,
1205
+ "eval_samples_per_second": 16.145,
1206
+ "eval_steps_per_second": 2.024,
1207
+ "step": 1140
1208
+ }
1209
+ ],
1210
+ "logging_steps": 20,
1211
+ "max_steps": 4150,
1212
+ "num_input_tokens_seen": 0,
1213
+ "num_train_epochs": 10,
1214
+ "save_steps": 20,
1215
+ "stateful_callbacks": {
1216
+ "TrainerControl": {
1217
+ "args": {
1218
+ "should_epoch_stop": false,
1219
+ "should_evaluate": false,
1220
+ "should_log": false,
1221
+ "should_save": true,
1222
+ "should_training_stop": false
1223
+ },
1224
+ "attributes": {}
1225
+ }
1226
+ },
1227
+ "total_flos": 1.1159542247710925e+17,
1228
+ "train_batch_size": 4,
1229
+ "trial_name": null,
1230
+ "trial_params": null
1231
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1160/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1160/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.03828026524568501,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "v_proj",
34
+ "up_proj",
35
+ "down_proj",
36
+ "o_proj",
37
+ "q_proj",
38
+ "k_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Estonian/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1160/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}