K2triinK commited on
Commit
cdaecd3
·
verified ·
1 Parent(s): 7e63740

Add files using upload-large-folder tool

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/README.md +58 -0
  2. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/README.md +58 -0
  3. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/README.md +209 -0
  4. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/adapter_config.json +46 -0
  5. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/chat_template.jinja +154 -0
  6. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/tokenizer_config.json +31 -0
  7. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/trainer_state.json +139 -0
  8. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/README.md +209 -0
  9. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/adapter_config.json +46 -0
  10. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/chat_template.jinja +154 -0
  11. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/tokenizer_config.json +31 -0
  12. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/trainer_state.json +1084 -0
  13. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/README.md +209 -0
  14. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/adapter_config.json +46 -0
  15. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/chat_template.jinja +154 -0
  16. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/tokenizer_config.json +31 -0
  17. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/trainer_state.json +1105 -0
  18. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/README.md +209 -0
  19. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/adapter_config.json +46 -0
  20. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/chat_template.jinja +154 -0
  21. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/tokenizer_config.json +31 -0
  22. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/trainer_state.json +1126 -0
  23. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/README.md +209 -0
  24. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/adapter_config.json +46 -0
  25. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/chat_template.jinja +154 -0
  26. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/tokenizer_config.json +31 -0
  27. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/trainer_state.json +1147 -0
  28. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/README.md +209 -0
  29. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/adapter_config.json +46 -0
  30. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/chat_template.jinja +154 -0
  31. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/tokenizer_config.json +31 -0
  32. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/trainer_state.json +1168 -0
  33. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/README.md +209 -0
  34. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/adapter_config.json +46 -0
  35. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/README.md +58 -0
  36. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1266/README.md +209 -0
  37. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1266/adapter_config.json +46 -0
  38. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1266/chat_template.jinja +154 -0
  39. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1266/tokenizer_config.json +31 -0
  40. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1266/trainer_state.json +317 -0
  41. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1688/README.md +209 -0
  42. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1688/adapter_config.json +46 -0
  43. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1688/chat_template.jinja +154 -0
  44. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1688/tokenizer_config.json +31 -0
  45. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1688/trainer_state.json +408 -0
  46. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-2110/README.md +209 -0
  47. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-2110/adapter_config.json +46 -0
  48. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-2110/chat_template.jinja +154 -0
  49. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-2110/tokenizer_config.json +31 -0
  50. productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-2110/trainer_state.json +509 -0
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/README.md ADDED
@@ -0,0 +1,58 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: transformers
4
+ model_name: Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1
5
+ tags:
6
+ - generated_from_trainer
7
+ - sft
8
+ - trl
9
+ licence: license
10
+ ---
11
+
12
+ # Model Card for Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1
13
+
14
+ This model is a fine-tuned version of [Qwen/Qwen3.5-4B-Base](https://huggingface.co/Qwen/Qwen3.5-4B-Base).
15
+ It has been trained using [TRL](https://github.com/huggingface/trl).
16
+
17
+ ## Quick start
18
+
19
+ ```python
20
+ from transformers import pipeline
21
+
22
+ question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?"
23
+ generator = pipeline("text-generation", model="None", device="cuda")
24
+ output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0]
25
+ print(output["generated_text"])
26
+ ```
27
+
28
+ ## Training procedure
29
+
30
+ [<img src="https://raw.githubusercontent.com/wandb/assets/main/wandb-github-badge-28.svg" alt="Visualize in Weights & Biases" width="150" height="24"/>](https://wandb.ai/katriin-kukk/Cross_lingual_morphological_generalization/runs/onphamd0)
31
+
32
+
33
+
34
+ This model was trained with SFT.
35
+
36
+ ### Framework versions
37
+
38
+ - TRL: 0.29.0
39
+ - Transformers: 5.5.4
40
+ - Pytorch: 2.10.0
41
+ - Datasets: 4.6.1
42
+ - Tokenizers: 0.22.2
43
+
44
+ ## Citations
45
+
46
+
47
+
48
+ Cite TRL as:
49
+
50
+ ```bibtex
51
+ @software{vonwerra2020trl,
52
+ title = {{TRL: Transformers Reinforcement Learning}},
53
+ author = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin},
54
+ license = {Apache-2.0},
55
+ url = {https://github.com/huggingface/trl},
56
+ year = {2020}
57
+ }
58
+ ```
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/README.md ADDED
@@ -0,0 +1,58 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: transformers
4
+ model_name: Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2
5
+ tags:
6
+ - generated_from_trainer
7
+ - sft
8
+ - trl
9
+ licence: license
10
+ ---
11
+
12
+ # Model Card for Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2
13
+
14
+ This model is a fine-tuned version of [Qwen/Qwen3.5-4B-Base](https://huggingface.co/Qwen/Qwen3.5-4B-Base).
15
+ It has been trained using [TRL](https://github.com/huggingface/trl).
16
+
17
+ ## Quick start
18
+
19
+ ```python
20
+ from transformers import pipeline
21
+
22
+ question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?"
23
+ generator = pipeline("text-generation", model="None", device="cuda")
24
+ output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0]
25
+ print(output["generated_text"])
26
+ ```
27
+
28
+ ## Training procedure
29
+
30
+ [<img src="https://raw.githubusercontent.com/wandb/assets/main/wandb-github-badge-28.svg" alt="Visualize in Weights & Biases" width="150" height="24"/>](https://wandb.ai/katriin-kukk/Cross_lingual_morphological_generalization/runs/7xpb8te3)
31
+
32
+
33
+
34
+ This model was trained with SFT.
35
+
36
+ ### Framework versions
37
+
38
+ - TRL: 0.29.0
39
+ - Transformers: 5.5.4
40
+ - Pytorch: 2.10.0
41
+ - Datasets: 4.6.1
42
+ - Tokenizers: 0.22.2
43
+
44
+ ## Citations
45
+
46
+
47
+
48
+ Cite TRL as:
49
+
50
+ ```bibtex
51
+ @software{vonwerra2020trl,
52
+ title = {{TRL: Transformers Reinforcement Learning}},
53
+ author = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin},
54
+ license = {Apache-2.0},
55
+ url = {https://github.com/huggingface/trl},
56
+ year = {2020}
57
+ }
58
+ ```
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.028265386974777595,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-100/trainer_state.json ADDED
@@ -0,0 +1,139 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.24906600249066002,
6
+ "eval_steps": 20,
7
+ "global_step": 100,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.955029806494713,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.020533561706543,
16
+ "learning_rate": 1.0694800202775147e-05,
17
+ "loss": 1.7107986450195312,
18
+ "mean_token_accuracy": 0.6487608112394809,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.3144892034835594,
25
+ "eval_loss": 1.1198534965515137,
26
+ "eval_mean_token_accuracy": 0.7460246955932572,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.0565,
29
+ "eval_samples_per_second": 15.794,
30
+ "eval_steps_per_second": 1.976,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.0063214391469955,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.572906494140625,
37
+ "learning_rate": 2.1952484626748985e-05,
38
+ "loss": 0.8663722991943359,
39
+ "mean_token_accuracy": 0.7779282338917255,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7921617945959402,
46
+ "eval_loss": 0.7062025666236877,
47
+ "eval_mean_token_accuracy": 0.8100443180910376,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.5189,
50
+ "eval_samples_per_second": 15.892,
51
+ "eval_steps_per_second": 1.988,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7682028576731682,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3003711700439453,
58
+ "learning_rate": 3.3210169050722824e-05,
59
+ "loss": 0.673183822631836,
60
+ "mean_token_accuracy": 0.8182129614055157,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7059133584762729,
67
+ "eval_loss": 0.6481946706771851,
68
+ "eval_mean_token_accuracy": 0.8227418761613757,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.5098,
71
+ "eval_samples_per_second": 15.894,
72
+ "eval_steps_per_second": 1.988,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7029960259795189,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.3664201498031616,
79
+ "learning_rate": 4.4467853474696664e-05,
80
+ "loss": 0.6354611873626709,
81
+ "mean_token_accuracy": 0.8243416830897331,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6867308004997498,
88
+ "eval_loss": 0.6179807186126709,
89
+ "eval_mean_token_accuracy": 0.8302594811417335,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.3969,
92
+ "eval_samples_per_second": 15.915,
93
+ "eval_steps_per_second": 1.991,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6764581337571144,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9815880656242371,
100
+ "learning_rate": 5.57255378986705e-05,
101
+ "loss": 0.5988658905029297,
102
+ "mean_token_accuracy": 0.8329168625175953,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6790881479202315,
109
+ "eval_loss": 0.5998476147651672,
110
+ "eval_mean_token_accuracy": 0.8318756420251935,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.6653,
113
+ "eval_samples_per_second": 15.866,
114
+ "eval_steps_per_second": 1.985,
115
+ "step": 100
116
+ }
117
+ ],
118
+ "logging_steps": 20,
119
+ "max_steps": 4020,
120
+ "num_input_tokens_seen": 0,
121
+ "num_train_epochs": 10,
122
+ "save_steps": 20,
123
+ "stateful_callbacks": {
124
+ "TrainerControl": {
125
+ "args": {
126
+ "should_epoch_stop": false,
127
+ "should_evaluate": false,
128
+ "should_log": false,
129
+ "should_save": true,
130
+ "should_training_stop": false
131
+ },
132
+ "attributes": {}
133
+ }
134
+ },
135
+ "total_flos": 9823576763965440.0,
136
+ "train_batch_size": 4,
137
+ "trial_name": null,
138
+ "trial_params": null
139
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.028265386974777595,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1000/trainer_state.json ADDED
@@ -0,0 +1,1084 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.488169364881694,
6
+ "eval_steps": 20,
7
+ "global_step": 1000,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.955029806494713,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.020533561706543,
16
+ "learning_rate": 1.0694800202775147e-05,
17
+ "loss": 1.7107986450195312,
18
+ "mean_token_accuracy": 0.6487608112394809,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.3144892034835594,
25
+ "eval_loss": 1.1198534965515137,
26
+ "eval_mean_token_accuracy": 0.7460246955932572,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.0565,
29
+ "eval_samples_per_second": 15.794,
30
+ "eval_steps_per_second": 1.976,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.0063214391469955,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.572906494140625,
37
+ "learning_rate": 2.1952484626748985e-05,
38
+ "loss": 0.8663722991943359,
39
+ "mean_token_accuracy": 0.7779282338917255,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7921617945959402,
46
+ "eval_loss": 0.7062025666236877,
47
+ "eval_mean_token_accuracy": 0.8100443180910376,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.5189,
50
+ "eval_samples_per_second": 15.892,
51
+ "eval_steps_per_second": 1.988,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7682028576731682,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3003711700439453,
58
+ "learning_rate": 3.3210169050722824e-05,
59
+ "loss": 0.673183822631836,
60
+ "mean_token_accuracy": 0.8182129614055157,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7059133584762729,
67
+ "eval_loss": 0.6481946706771851,
68
+ "eval_mean_token_accuracy": 0.8227418761613757,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.5098,
71
+ "eval_samples_per_second": 15.894,
72
+ "eval_steps_per_second": 1.988,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7029960259795189,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.3664201498031616,
79
+ "learning_rate": 4.4467853474696664e-05,
80
+ "loss": 0.6354611873626709,
81
+ "mean_token_accuracy": 0.8243416830897331,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6867308004997498,
88
+ "eval_loss": 0.6179807186126709,
89
+ "eval_mean_token_accuracy": 0.8302594811417335,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.3969,
92
+ "eval_samples_per_second": 15.915,
93
+ "eval_steps_per_second": 1.991,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6764581337571144,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9815880656242371,
100
+ "learning_rate": 5.57255378986705e-05,
101
+ "loss": 0.5988658905029297,
102
+ "mean_token_accuracy": 0.8329168625175953,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6790881479202315,
109
+ "eval_loss": 0.5998476147651672,
110
+ "eval_mean_token_accuracy": 0.8318756420251935,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.6653,
113
+ "eval_samples_per_second": 15.866,
114
+ "eval_steps_per_second": 1.985,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6744543805718421,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.932099461555481,
121
+ "learning_rate": 6.698322232264434e-05,
122
+ "loss": 0.5991750717163086,
123
+ "mean_token_accuracy": 0.8304223112761975,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.6813044282932614,
130
+ "eval_loss": 0.5922021269798279,
131
+ "eval_mean_token_accuracy": 0.8346439617317777,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.1551,
134
+ "eval_samples_per_second": 15.96,
135
+ "eval_steps_per_second": 1.996,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6663189359009266,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9528499841690063,
142
+ "learning_rate": 7.824090674661818e-05,
143
+ "loss": 0.5891091346740722,
144
+ "mean_token_accuracy": 0.832152470946312,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6398407208711602,
151
+ "eval_loss": 0.5859636664390564,
152
+ "eval_mean_token_accuracy": 0.8372074996316156,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.2706,
155
+ "eval_samples_per_second": 15.938,
156
+ "eval_steps_per_second": 1.994,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.64859763905406,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8468204140663147,
163
+ "learning_rate": 8.949859117059201e-05,
164
+ "loss": 0.569426441192627,
165
+ "mean_token_accuracy": 0.8401990942656994,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6381674285891444,
172
+ "eval_loss": 0.5744525790214539,
173
+ "eval_mean_token_accuracy": 0.838626817908398,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.2848,
176
+ "eval_samples_per_second": 15.936,
177
+ "eval_steps_per_second": 1.993,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6432608783245086,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8765804767608643,
184
+ "learning_rate": 0.00010075627559456587,
185
+ "loss": 0.5687318801879883,
186
+ "mean_token_accuracy": 0.839249350130558,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6047098288355872,
193
+ "eval_loss": 0.5679298043251038,
194
+ "eval_mean_token_accuracy": 0.8410577181466791,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.5879,
197
+ "eval_samples_per_second": 15.88,
198
+ "eval_steps_per_second": 1.986,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6440276011824608,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9576020240783691,
205
+ "learning_rate": 0.00011201396001853971,
206
+ "loss": 0.5828506469726562,
207
+ "mean_token_accuracy": 0.837553184479475,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6161119360909906,
214
+ "eval_loss": 0.5702911615371704,
215
+ "eval_mean_token_accuracy": 0.8407089398350827,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3341,
218
+ "eval_samples_per_second": 15.926,
219
+ "eval_steps_per_second": 1.992,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6377195850014686,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7212373614311218,
226
+ "learning_rate": 0.00012327164444251353,
227
+ "loss": 0.5702451229095459,
228
+ "mean_token_accuracy": 0.8397969007492065,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6080108886194784,
235
+ "eval_loss": 0.5633499622344971,
236
+ "eval_mean_token_accuracy": 0.8396634854549585,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.4945,
239
+ "eval_samples_per_second": 15.897,
240
+ "eval_steps_per_second": 1.989,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6287345830351114,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.848779022693634,
247
+ "learning_rate": 0.00013452932886648739,
248
+ "loss": 0.5506546020507812,
249
+ "mean_token_accuracy": 0.8438881888985634,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6307531505130058,
256
+ "eval_loss": 0.5573338270187378,
257
+ "eval_mean_token_accuracy": 0.8431362606758295,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.3535,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.992,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6223786748945713,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.7316951751708984,
268
+ "learning_rate": 0.0001457870132904612,
269
+ "loss": 0.5495625972747803,
270
+ "mean_token_accuracy": 0.8440376669168472,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.623454462476941,
277
+ "eval_loss": 0.5619264245033264,
278
+ "eval_mean_token_accuracy": 0.8431175777385401,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.2008,
281
+ "eval_samples_per_second": 15.951,
282
+ "eval_steps_per_second": 1.995,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6281675305217505,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7639564871788025,
289
+ "learning_rate": 0.00015704469771443506,
290
+ "loss": 0.5604369163513183,
291
+ "mean_token_accuracy": 0.8401600055396556,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.63416675980701,
298
+ "eval_loss": 0.5612760782241821,
299
+ "eval_mean_token_accuracy": 0.842435666294985,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.25,
302
+ "eval_samples_per_second": 15.942,
303
+ "eval_steps_per_second": 1.994,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6427909277379513,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6475813388824463,
310
+ "learning_rate": 0.0001683023821384089,
311
+ "loss": 0.573763370513916,
312
+ "mean_token_accuracy": 0.8370340794324875,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6231539840268534,
319
+ "eval_loss": 0.5566866397857666,
320
+ "eval_mean_token_accuracy": 0.844177934319474,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.4858,
323
+ "eval_samples_per_second": 15.899,
324
+ "eval_steps_per_second": 1.989,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6226776849478484,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.8886699676513672,
331
+ "learning_rate": 0.00017956006656238274,
332
+ "loss": 0.558210802078247,
333
+ "mean_token_accuracy": 0.84083157107234,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6066981683983359,
340
+ "eval_loss": 0.5585207939147949,
341
+ "eval_mean_token_accuracy": 0.8423153311014175,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.3463,
344
+ "eval_samples_per_second": 15.924,
345
+ "eval_steps_per_second": 1.992,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6249004438519478,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8791211843490601,
352
+ "learning_rate": 0.00019081775098635657,
353
+ "loss": 0.5603597164154053,
354
+ "mean_token_accuracy": 0.8420463085174561,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6082247584018596,
361
+ "eval_loss": 0.5616299510002136,
362
+ "eval_mean_token_accuracy": 0.8431286801432454,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.1253,
365
+ "eval_samples_per_second": 15.965,
366
+ "eval_steps_per_second": 1.997,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6362396612763405,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.8606319427490234,
373
+ "learning_rate": 0.0002020754354103304,
374
+ "loss": 0.5735773563385009,
375
+ "mean_token_accuracy": 0.8371490836143494,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6492362072648004,
382
+ "eval_loss": 0.5646467804908752,
383
+ "eval_mean_token_accuracy": 0.8415517574825953,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.3351,
386
+ "eval_samples_per_second": 15.926,
387
+ "eval_steps_per_second": 1.992,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.638665035739541,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.7773950099945068,
394
+ "learning_rate": 0.00021333311983430425,
395
+ "loss": 0.5820859909057617,
396
+ "mean_token_accuracy": 0.8372561208903789,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6434498637221581,
403
+ "eval_loss": 0.5645168423652649,
404
+ "eval_mean_token_accuracy": 0.8420382481674815,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.1216,
407
+ "eval_samples_per_second": 15.966,
408
+ "eval_steps_per_second": 1.997,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6316851265728474,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 1.6120579242706299,
415
+ "learning_rate": 0.00022459080425827807,
416
+ "loss": 0.5637502670288086,
417
+ "mean_token_accuracy": 0.8386227294802666,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6469012776086497,
424
+ "eval_loss": 0.5758090615272522,
425
+ "eval_mean_token_accuracy": 0.8397158470957778,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.6139,
428
+ "eval_samples_per_second": 15.875,
429
+ "eval_steps_per_second": 1.986,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5894816922835815,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 1.1616325378417969,
436
+ "learning_rate": 0.00022626713048053178,
437
+ "loss": 0.5316025257110596,
438
+ "mean_token_accuracy": 0.8466163017810919,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5860798164855602,
445
+ "eval_loss": 0.5777581930160522,
446
+ "eval_mean_token_accuracy": 0.8396938103576039,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.1449,
449
+ "eval_samples_per_second": 15.961,
450
+ "eval_steps_per_second": 1.997,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5818420693278312,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.7999453544616699,
457
+ "learning_rate": 0.00022622107023288778,
458
+ "loss": 0.5221010208129883,
459
+ "mean_token_accuracy": 0.8474301159381866,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5783926014636838,
466
+ "eval_loss": 0.5700300931930542,
467
+ "eval_mean_token_accuracy": 0.8430753537388735,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.5308,
470
+ "eval_samples_per_second": 15.89,
471
+ "eval_steps_per_second": 1.988,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5612493887543678,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 1.015687346458435,
478
+ "learning_rate": 0.00022614090619491568,
479
+ "loss": 0.5084867000579834,
480
+ "mean_token_accuracy": 0.8495561093091964,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5841563874205877,
487
+ "eval_loss": 0.5693665742874146,
488
+ "eval_mean_token_accuracy": 0.8427817298229351,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.5256,
491
+ "eval_samples_per_second": 15.891,
492
+ "eval_steps_per_second": 1.988,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5828216474503278,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 1.9750930070877075,
499
+ "learning_rate": 0.00022602666254299594,
500
+ "loss": 0.5180017948150635,
501
+ "mean_token_accuracy": 0.8515685826539994,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5806607044366903,
508
+ "eval_loss": 0.5804352760314941,
509
+ "eval_mean_token_accuracy": 0.8413014668364858,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.1199,
512
+ "eval_samples_per_second": 15.966,
513
+ "eval_steps_per_second": 1.997,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5926914308220148,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 0.8917353749275208,
520
+ "learning_rate": 0.0002258783737314558,
521
+ "loss": 0.528910779953003,
522
+ "mean_token_accuracy": 0.8486074328422546,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5593361884009006,
529
+ "eval_loss": 0.5675153732299805,
530
+ "eval_mean_token_accuracy": 0.8433507802181466,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7289,
533
+ "eval_samples_per_second": 15.854,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5865630559623242,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7482362985610962,
541
+ "learning_rate": 0.00022569608448217823,
542
+ "loss": 0.5250466823577881,
543
+ "mean_token_accuracy": 0.8477916084229946,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.543057840230853,
550
+ "eval_loss": 0.5671008229255676,
551
+ "eval_mean_token_accuracy": 0.8428726016088973,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.3403,
554
+ "eval_samples_per_second": 15.925,
555
+ "eval_steps_per_second": 1.992,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5870206747204065,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9473814964294434,
562
+ "learning_rate": 0.00022547984977111448,
563
+ "loss": 0.5252370834350586,
564
+ "mean_token_accuracy": 0.8468369916081429,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.590982622878496,
571
+ "eval_loss": 0.5676343441009521,
572
+ "eval_mean_token_accuracy": 0.8429348746011424,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.5168,
575
+ "eval_samples_per_second": 15.893,
576
+ "eval_steps_per_second": 1.988,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5785854265093804,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.9353351593017578,
583
+ "learning_rate": 0.0002252297348117042,
584
+ "loss": 0.5304938316345215,
585
+ "mean_token_accuracy": 0.8463383808732032,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6099918867612995,
592
+ "eval_loss": 0.5620437860488892,
593
+ "eval_mean_token_accuracy": 0.8430728347495545,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.7741,
596
+ "eval_samples_per_second": 15.846,
597
+ "eval_steps_per_second": 1.982,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5768801040947438,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.9198738932609558,
604
+ "learning_rate": 0.0002249458150352077,
605
+ "loss": 0.520513391494751,
606
+ "mean_token_accuracy": 0.8487689301371575,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.6349420670506566,
613
+ "eval_loss": 0.5645340085029602,
614
+ "eval_mean_token_accuracy": 0.8447844597489335,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.3257,
617
+ "eval_samples_per_second": 15.928,
618
+ "eval_steps_per_second": 1.992,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5822233572602272,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.832811176776886,
625
+ "learning_rate": 0.0002246281760679571,
626
+ "loss": 0.5295282363891601,
627
+ "mean_token_accuracy": 0.8504064798355102,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5829724387027496,
634
+ "eval_loss": 0.5612193942070007,
635
+ "eval_mean_token_accuracy": 0.8449643853791925,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6617,
638
+ "eval_samples_per_second": 15.866,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.571855777129531,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.7665547728538513,
646
+ "learning_rate": 0.00022427691370553263,
647
+ "loss": 0.5187931060791016,
648
+ "mean_token_accuracy": 0.8534420043230057,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5623592240519302,
655
+ "eval_loss": 0.5575760006904602,
656
+ "eval_mean_token_accuracy": 0.8468210229346919,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.6324,
659
+ "eval_samples_per_second": 15.872,
660
+ "eval_steps_per_second": 1.985,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5740394659340382,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6500429511070251,
667
+ "learning_rate": 0.00022389213388387174,
668
+ "loss": 0.5283198833465577,
669
+ "mean_token_accuracy": 0.8502798482775689,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5548852207355721,
676
+ "eval_loss": 0.5561797022819519,
677
+ "eval_mean_token_accuracy": 0.8452786498291548,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.5205,
680
+ "eval_samples_per_second": 15.892,
681
+ "eval_steps_per_second": 1.988,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.6020145989954472,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.7056867480278015,
688
+ "learning_rate": 0.00022347395264732053,
689
+ "loss": 0.5400049209594726,
690
+ "mean_token_accuracy": 0.8447613954544068,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5618055154417836,
697
+ "eval_loss": 0.556106686592102,
698
+ "eval_mean_token_accuracy": 0.8465680112672407,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.2971,
701
+ "eval_samples_per_second": 15.933,
702
+ "eval_steps_per_second": 1.993,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5665927153080702,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.5987663865089417,
709
+ "learning_rate": 0.00022302249611363625,
710
+ "loss": 0.5143643856048584,
711
+ "mean_token_accuracy": 0.8529589556157589,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.568248552118623,
718
+ "eval_loss": 0.5476346015930176,
719
+ "eval_mean_token_accuracy": 0.8476775434128073,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.9583,
722
+ "eval_samples_per_second": 15.812,
723
+ "eval_steps_per_second": 1.978,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5673687808215618,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.735261857509613,
730
+ "learning_rate": 0.00022253790043595193,
731
+ "loss": 0.509885597229004,
732
+ "mean_token_accuracy": 0.8537046857178211,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5616967284748721,
739
+ "eval_loss": 0.5439274311065674,
740
+ "eval_mean_token_accuracy": 0.8488946217437123,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.0604,
743
+ "eval_samples_per_second": 15.977,
744
+ "eval_steps_per_second": 1.999,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.5529541682451964,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7014835476875305,
751
+ "learning_rate": 0.00022202031176171442,
752
+ "loss": 0.5078992366790771,
753
+ "mean_token_accuracy": 0.8525233261287213,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5827173320359962,
760
+ "eval_loss": 0.5419450402259827,
761
+ "eval_mean_token_accuracy": 0.8477318609176681,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 85.2984,
764
+ "eval_samples_per_second": 16.12,
765
+ "eval_steps_per_second": 2.016,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5755720350891351,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.705613911151886,
772
+ "learning_rate": 0.00022146988618860824,
773
+ "loss": 0.5181350708007812,
774
+ "mean_token_accuracy": 0.8467609457671642,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5743971356125765,
781
+ "eval_loss": 0.5415896773338318,
782
+ "eval_mean_token_accuracy": 0.847328585940738,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 85.5602,
785
+ "eval_samples_per_second": 16.071,
786
+ "eval_steps_per_second": 2.01,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.561330484598875,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.6722865700721741,
793
+ "learning_rate": 0.0002208867897174789,
794
+ "loss": 0.499837589263916,
795
+ "mean_token_accuracy": 0.8518734864890576,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.5865232653396074,
802
+ "eval_loss": 0.5437926650047302,
803
+ "eval_mean_token_accuracy": 0.8450997017843779,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4116,
806
+ "eval_samples_per_second": 15.912,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.547389242425561,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.7935577034950256,
814
+ "learning_rate": 0.00022027119820226907,
815
+ "loss": 0.4977591514587402,
816
+ "mean_token_accuracy": 0.8539491161704064,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.5290903090391048,
823
+ "eval_loss": 0.5409526824951172,
824
+ "eval_mean_token_accuracy": 0.8497545698354411,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.7262,
827
+ "eval_samples_per_second": 15.854,
828
+ "eval_steps_per_second": 1.983,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5687909748405218,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6180546283721924,
835
+ "learning_rate": 0.00021962329729698345,
836
+ "loss": 0.5109643459320068,
837
+ "mean_token_accuracy": 0.8521598495543004,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.5503541858390321,
844
+ "eval_loss": 0.5361555218696594,
845
+ "eval_mean_token_accuracy": 0.8510884285666221,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.3339,
848
+ "eval_samples_per_second": 15.927,
849
+ "eval_steps_per_second": 1.992,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4739728841261986,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.8058829307556152,
856
+ "learning_rate": 0.0002189432823996982,
857
+ "loss": 0.4204097747802734,
858
+ "mean_token_accuracy": 0.8728981889211215,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.5077334992414297,
865
+ "eval_loss": 0.5531114339828491,
866
+ "eval_mean_token_accuracy": 0.8489257208136625,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.4801,
869
+ "eval_samples_per_second": 15.9,
870
+ "eval_steps_per_second": 1.989,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.4594309840351343,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.6906896829605103,
877
+ "learning_rate": 0.0002182313585936314,
878
+ "loss": 0.4071959495544434,
879
+ "mean_token_accuracy": 0.8732857562601566,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.49850136994622474,
886
+ "eval_loss": 0.5486204624176025,
887
+ "eval_mean_token_accuracy": 0.8507991450470548,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.3364,
890
+ "eval_samples_per_second": 15.926,
891
+ "eval_steps_per_second": 1.992,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4881629109382629,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6343470215797424,
898
+ "learning_rate": 0.0002174877405852928,
899
+ "loss": 0.41669540405273436,
900
+ "mean_token_accuracy": 0.8711295068264008,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.49155513924914734,
907
+ "eval_loss": 0.555109441280365,
908
+ "eval_mean_token_accuracy": 0.8496399400539176,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 86.3295,
911
+ "eval_samples_per_second": 15.927,
912
+ "eval_steps_per_second": 1.992,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.4648668970912695,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.8014165163040161,
919
+ "learning_rate": 0.00021671265263973133,
920
+ "loss": 0.4110250473022461,
921
+ "mean_token_accuracy": 0.8754166305065155,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.4909258722219356,
928
+ "eval_loss": 0.5539511442184448,
929
+ "eval_mean_token_accuracy": 0.8492401502160138,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.3468,
932
+ "eval_samples_per_second": 15.924,
933
+ "eval_steps_per_second": 1.992,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4824485514312983,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6665191054344177,
940
+ "learning_rate": 0.00021590632851289967,
941
+ "loss": 0.4181404113769531,
942
+ "mean_token_accuracy": 0.8726993151009083,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.4986876940657926,
949
+ "eval_loss": 0.547695517539978,
950
+ "eval_mean_token_accuracy": 0.8501384708770486,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.3838,
953
+ "eval_samples_per_second": 15.917,
954
+ "eval_steps_per_second": 1.991,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.4751896943897009,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.81158047914505,
961
+ "learning_rate": 0.00021506901138115678,
962
+ "loss": 0.40689678192138673,
963
+ "mean_token_accuracy": 0.8745221219956875,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.507153491121392,
970
+ "eval_loss": 0.5501641631126404,
971
+ "eval_mean_token_accuracy": 0.8495670116918032,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.0912,
974
+ "eval_samples_per_second": 15.971,
975
+ "eval_steps_per_second": 1.998,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4873133715242147,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7218056321144104,
982
+ "learning_rate": 0.0002142009537679292,
983
+ "loss": 0.42701358795166017,
984
+ "mean_token_accuracy": 0.8695114746689796,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5202612736543943,
991
+ "eval_loss": 0.5491839051246643,
992
+ "eval_mean_token_accuracy": 0.8494071208460386,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.1142,
995
+ "eval_samples_per_second": 15.967,
996
+ "eval_steps_per_second": 1.997,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.4762951169162989,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 0.7194424867630005,
1003
+ "learning_rate": 0.0002133024174675534,
1004
+ "loss": 0.42299847602844237,
1005
+ "mean_token_accuracy": 0.8709790132939815,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4899340462546016,
1012
+ "eval_loss": 0.5522511601448059,
1013
+ "eval_mean_token_accuracy": 0.8492208258357159,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.463,
1016
+ "eval_samples_per_second": 15.903,
1017
+ "eval_steps_per_second": 1.989,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.49650347977876663,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.8406022787094116,
1024
+ "learning_rate": 0.0002123736734663221,
1025
+ "loss": 0.4275330066680908,
1026
+ "mean_token_accuracy": 0.8670595556497573,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.49691385654515996,
1033
+ "eval_loss": 0.5491269826889038,
1034
+ "eval_mean_token_accuracy": 0.850309816210769,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.17,
1037
+ "eval_samples_per_second": 15.957,
1038
+ "eval_steps_per_second": 1.996,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.48843890577554705,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.9082473516464233,
1045
+ "learning_rate": 0.00021141500186075868,
1046
+ "loss": 0.4309722423553467,
1047
+ "mean_token_accuracy": 0.8686766296625137,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5543508351195691,
1054
+ "eval_loss": 0.5478800535202026,
1055
+ "eval_mean_token_accuracy": 0.8478029522784921,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.3835,
1058
+ "eval_samples_per_second": 15.917,
1059
+ "eval_steps_per_second": 1.991,
1060
+ "step": 1000
1061
+ }
1062
+ ],
1063
+ "logging_steps": 20,
1064
+ "max_steps": 4020,
1065
+ "num_input_tokens_seen": 0,
1066
+ "num_train_epochs": 10,
1067
+ "save_steps": 20,
1068
+ "stateful_callbacks": {
1069
+ "TrainerControl": {
1070
+ "args": {
1071
+ "should_epoch_stop": false,
1072
+ "should_evaluate": false,
1073
+ "should_log": false,
1074
+ "should_save": true,
1075
+ "should_training_stop": false
1076
+ },
1077
+ "attributes": {}
1078
+ }
1079
+ },
1080
+ "total_flos": 9.859037950771814e+16,
1081
+ "train_batch_size": 4,
1082
+ "trial_name": null,
1083
+ "trial_params": null
1084
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.028265386974777595,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1020/trainer_state.json ADDED
@@ -0,0 +1,1105 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.5379825653798256,
6
+ "eval_steps": 20,
7
+ "global_step": 1020,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.955029806494713,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.020533561706543,
16
+ "learning_rate": 1.0694800202775147e-05,
17
+ "loss": 1.7107986450195312,
18
+ "mean_token_accuracy": 0.6487608112394809,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.3144892034835594,
25
+ "eval_loss": 1.1198534965515137,
26
+ "eval_mean_token_accuracy": 0.7460246955932572,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.0565,
29
+ "eval_samples_per_second": 15.794,
30
+ "eval_steps_per_second": 1.976,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.0063214391469955,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.572906494140625,
37
+ "learning_rate": 2.1952484626748985e-05,
38
+ "loss": 0.8663722991943359,
39
+ "mean_token_accuracy": 0.7779282338917255,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7921617945959402,
46
+ "eval_loss": 0.7062025666236877,
47
+ "eval_mean_token_accuracy": 0.8100443180910376,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.5189,
50
+ "eval_samples_per_second": 15.892,
51
+ "eval_steps_per_second": 1.988,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7682028576731682,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3003711700439453,
58
+ "learning_rate": 3.3210169050722824e-05,
59
+ "loss": 0.673183822631836,
60
+ "mean_token_accuracy": 0.8182129614055157,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7059133584762729,
67
+ "eval_loss": 0.6481946706771851,
68
+ "eval_mean_token_accuracy": 0.8227418761613757,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.5098,
71
+ "eval_samples_per_second": 15.894,
72
+ "eval_steps_per_second": 1.988,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7029960259795189,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.3664201498031616,
79
+ "learning_rate": 4.4467853474696664e-05,
80
+ "loss": 0.6354611873626709,
81
+ "mean_token_accuracy": 0.8243416830897331,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6867308004997498,
88
+ "eval_loss": 0.6179807186126709,
89
+ "eval_mean_token_accuracy": 0.8302594811417335,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.3969,
92
+ "eval_samples_per_second": 15.915,
93
+ "eval_steps_per_second": 1.991,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6764581337571144,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9815880656242371,
100
+ "learning_rate": 5.57255378986705e-05,
101
+ "loss": 0.5988658905029297,
102
+ "mean_token_accuracy": 0.8329168625175953,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6790881479202315,
109
+ "eval_loss": 0.5998476147651672,
110
+ "eval_mean_token_accuracy": 0.8318756420251935,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.6653,
113
+ "eval_samples_per_second": 15.866,
114
+ "eval_steps_per_second": 1.985,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6744543805718421,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.932099461555481,
121
+ "learning_rate": 6.698322232264434e-05,
122
+ "loss": 0.5991750717163086,
123
+ "mean_token_accuracy": 0.8304223112761975,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.6813044282932614,
130
+ "eval_loss": 0.5922021269798279,
131
+ "eval_mean_token_accuracy": 0.8346439617317777,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.1551,
134
+ "eval_samples_per_second": 15.96,
135
+ "eval_steps_per_second": 1.996,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6663189359009266,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9528499841690063,
142
+ "learning_rate": 7.824090674661818e-05,
143
+ "loss": 0.5891091346740722,
144
+ "mean_token_accuracy": 0.832152470946312,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6398407208711602,
151
+ "eval_loss": 0.5859636664390564,
152
+ "eval_mean_token_accuracy": 0.8372074996316156,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.2706,
155
+ "eval_samples_per_second": 15.938,
156
+ "eval_steps_per_second": 1.994,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.64859763905406,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8468204140663147,
163
+ "learning_rate": 8.949859117059201e-05,
164
+ "loss": 0.569426441192627,
165
+ "mean_token_accuracy": 0.8401990942656994,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6381674285891444,
172
+ "eval_loss": 0.5744525790214539,
173
+ "eval_mean_token_accuracy": 0.838626817908398,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.2848,
176
+ "eval_samples_per_second": 15.936,
177
+ "eval_steps_per_second": 1.993,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6432608783245086,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8765804767608643,
184
+ "learning_rate": 0.00010075627559456587,
185
+ "loss": 0.5687318801879883,
186
+ "mean_token_accuracy": 0.839249350130558,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6047098288355872,
193
+ "eval_loss": 0.5679298043251038,
194
+ "eval_mean_token_accuracy": 0.8410577181466791,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.5879,
197
+ "eval_samples_per_second": 15.88,
198
+ "eval_steps_per_second": 1.986,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6440276011824608,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9576020240783691,
205
+ "learning_rate": 0.00011201396001853971,
206
+ "loss": 0.5828506469726562,
207
+ "mean_token_accuracy": 0.837553184479475,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6161119360909906,
214
+ "eval_loss": 0.5702911615371704,
215
+ "eval_mean_token_accuracy": 0.8407089398350827,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3341,
218
+ "eval_samples_per_second": 15.926,
219
+ "eval_steps_per_second": 1.992,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6377195850014686,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7212373614311218,
226
+ "learning_rate": 0.00012327164444251353,
227
+ "loss": 0.5702451229095459,
228
+ "mean_token_accuracy": 0.8397969007492065,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6080108886194784,
235
+ "eval_loss": 0.5633499622344971,
236
+ "eval_mean_token_accuracy": 0.8396634854549585,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.4945,
239
+ "eval_samples_per_second": 15.897,
240
+ "eval_steps_per_second": 1.989,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6287345830351114,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.848779022693634,
247
+ "learning_rate": 0.00013452932886648739,
248
+ "loss": 0.5506546020507812,
249
+ "mean_token_accuracy": 0.8438881888985634,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6307531505130058,
256
+ "eval_loss": 0.5573338270187378,
257
+ "eval_mean_token_accuracy": 0.8431362606758295,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.3535,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.992,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6223786748945713,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.7316951751708984,
268
+ "learning_rate": 0.0001457870132904612,
269
+ "loss": 0.5495625972747803,
270
+ "mean_token_accuracy": 0.8440376669168472,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.623454462476941,
277
+ "eval_loss": 0.5619264245033264,
278
+ "eval_mean_token_accuracy": 0.8431175777385401,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.2008,
281
+ "eval_samples_per_second": 15.951,
282
+ "eval_steps_per_second": 1.995,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6281675305217505,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7639564871788025,
289
+ "learning_rate": 0.00015704469771443506,
290
+ "loss": 0.5604369163513183,
291
+ "mean_token_accuracy": 0.8401600055396556,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.63416675980701,
298
+ "eval_loss": 0.5612760782241821,
299
+ "eval_mean_token_accuracy": 0.842435666294985,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.25,
302
+ "eval_samples_per_second": 15.942,
303
+ "eval_steps_per_second": 1.994,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6427909277379513,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6475813388824463,
310
+ "learning_rate": 0.0001683023821384089,
311
+ "loss": 0.573763370513916,
312
+ "mean_token_accuracy": 0.8370340794324875,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6231539840268534,
319
+ "eval_loss": 0.5566866397857666,
320
+ "eval_mean_token_accuracy": 0.844177934319474,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.4858,
323
+ "eval_samples_per_second": 15.899,
324
+ "eval_steps_per_second": 1.989,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6226776849478484,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.8886699676513672,
331
+ "learning_rate": 0.00017956006656238274,
332
+ "loss": 0.558210802078247,
333
+ "mean_token_accuracy": 0.84083157107234,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6066981683983359,
340
+ "eval_loss": 0.5585207939147949,
341
+ "eval_mean_token_accuracy": 0.8423153311014175,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.3463,
344
+ "eval_samples_per_second": 15.924,
345
+ "eval_steps_per_second": 1.992,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6249004438519478,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8791211843490601,
352
+ "learning_rate": 0.00019081775098635657,
353
+ "loss": 0.5603597164154053,
354
+ "mean_token_accuracy": 0.8420463085174561,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6082247584018596,
361
+ "eval_loss": 0.5616299510002136,
362
+ "eval_mean_token_accuracy": 0.8431286801432454,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.1253,
365
+ "eval_samples_per_second": 15.965,
366
+ "eval_steps_per_second": 1.997,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6362396612763405,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.8606319427490234,
373
+ "learning_rate": 0.0002020754354103304,
374
+ "loss": 0.5735773563385009,
375
+ "mean_token_accuracy": 0.8371490836143494,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6492362072648004,
382
+ "eval_loss": 0.5646467804908752,
383
+ "eval_mean_token_accuracy": 0.8415517574825953,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.3351,
386
+ "eval_samples_per_second": 15.926,
387
+ "eval_steps_per_second": 1.992,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.638665035739541,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.7773950099945068,
394
+ "learning_rate": 0.00021333311983430425,
395
+ "loss": 0.5820859909057617,
396
+ "mean_token_accuracy": 0.8372561208903789,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6434498637221581,
403
+ "eval_loss": 0.5645168423652649,
404
+ "eval_mean_token_accuracy": 0.8420382481674815,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.1216,
407
+ "eval_samples_per_second": 15.966,
408
+ "eval_steps_per_second": 1.997,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6316851265728474,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 1.6120579242706299,
415
+ "learning_rate": 0.00022459080425827807,
416
+ "loss": 0.5637502670288086,
417
+ "mean_token_accuracy": 0.8386227294802666,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6469012776086497,
424
+ "eval_loss": 0.5758090615272522,
425
+ "eval_mean_token_accuracy": 0.8397158470957778,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.6139,
428
+ "eval_samples_per_second": 15.875,
429
+ "eval_steps_per_second": 1.986,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5894816922835815,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 1.1616325378417969,
436
+ "learning_rate": 0.00022626713048053178,
437
+ "loss": 0.5316025257110596,
438
+ "mean_token_accuracy": 0.8466163017810919,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5860798164855602,
445
+ "eval_loss": 0.5777581930160522,
446
+ "eval_mean_token_accuracy": 0.8396938103576039,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.1449,
449
+ "eval_samples_per_second": 15.961,
450
+ "eval_steps_per_second": 1.997,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5818420693278312,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.7999453544616699,
457
+ "learning_rate": 0.00022622107023288778,
458
+ "loss": 0.5221010208129883,
459
+ "mean_token_accuracy": 0.8474301159381866,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5783926014636838,
466
+ "eval_loss": 0.5700300931930542,
467
+ "eval_mean_token_accuracy": 0.8430753537388735,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.5308,
470
+ "eval_samples_per_second": 15.89,
471
+ "eval_steps_per_second": 1.988,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5612493887543678,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 1.015687346458435,
478
+ "learning_rate": 0.00022614090619491568,
479
+ "loss": 0.5084867000579834,
480
+ "mean_token_accuracy": 0.8495561093091964,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5841563874205877,
487
+ "eval_loss": 0.5693665742874146,
488
+ "eval_mean_token_accuracy": 0.8427817298229351,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.5256,
491
+ "eval_samples_per_second": 15.891,
492
+ "eval_steps_per_second": 1.988,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5828216474503278,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 1.9750930070877075,
499
+ "learning_rate": 0.00022602666254299594,
500
+ "loss": 0.5180017948150635,
501
+ "mean_token_accuracy": 0.8515685826539994,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5806607044366903,
508
+ "eval_loss": 0.5804352760314941,
509
+ "eval_mean_token_accuracy": 0.8413014668364858,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.1199,
512
+ "eval_samples_per_second": 15.966,
513
+ "eval_steps_per_second": 1.997,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5926914308220148,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 0.8917353749275208,
520
+ "learning_rate": 0.0002258783737314558,
521
+ "loss": 0.528910779953003,
522
+ "mean_token_accuracy": 0.8486074328422546,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5593361884009006,
529
+ "eval_loss": 0.5675153732299805,
530
+ "eval_mean_token_accuracy": 0.8433507802181466,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7289,
533
+ "eval_samples_per_second": 15.854,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5865630559623242,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7482362985610962,
541
+ "learning_rate": 0.00022569608448217823,
542
+ "loss": 0.5250466823577881,
543
+ "mean_token_accuracy": 0.8477916084229946,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.543057840230853,
550
+ "eval_loss": 0.5671008229255676,
551
+ "eval_mean_token_accuracy": 0.8428726016088973,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.3403,
554
+ "eval_samples_per_second": 15.925,
555
+ "eval_steps_per_second": 1.992,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5870206747204065,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9473814964294434,
562
+ "learning_rate": 0.00022547984977111448,
563
+ "loss": 0.5252370834350586,
564
+ "mean_token_accuracy": 0.8468369916081429,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.590982622878496,
571
+ "eval_loss": 0.5676343441009521,
572
+ "eval_mean_token_accuracy": 0.8429348746011424,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.5168,
575
+ "eval_samples_per_second": 15.893,
576
+ "eval_steps_per_second": 1.988,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5785854265093804,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.9353351593017578,
583
+ "learning_rate": 0.0002252297348117042,
584
+ "loss": 0.5304938316345215,
585
+ "mean_token_accuracy": 0.8463383808732032,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6099918867612995,
592
+ "eval_loss": 0.5620437860488892,
593
+ "eval_mean_token_accuracy": 0.8430728347495545,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.7741,
596
+ "eval_samples_per_second": 15.846,
597
+ "eval_steps_per_second": 1.982,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5768801040947438,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.9198738932609558,
604
+ "learning_rate": 0.0002249458150352077,
605
+ "loss": 0.520513391494751,
606
+ "mean_token_accuracy": 0.8487689301371575,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.6349420670506566,
613
+ "eval_loss": 0.5645340085029602,
614
+ "eval_mean_token_accuracy": 0.8447844597489335,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.3257,
617
+ "eval_samples_per_second": 15.928,
618
+ "eval_steps_per_second": 1.992,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5822233572602272,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.832811176776886,
625
+ "learning_rate": 0.0002246281760679571,
626
+ "loss": 0.5295282363891601,
627
+ "mean_token_accuracy": 0.8504064798355102,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5829724387027496,
634
+ "eval_loss": 0.5612193942070007,
635
+ "eval_mean_token_accuracy": 0.8449643853791925,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6617,
638
+ "eval_samples_per_second": 15.866,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.571855777129531,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.7665547728538513,
646
+ "learning_rate": 0.00022427691370553263,
647
+ "loss": 0.5187931060791016,
648
+ "mean_token_accuracy": 0.8534420043230057,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5623592240519302,
655
+ "eval_loss": 0.5575760006904602,
656
+ "eval_mean_token_accuracy": 0.8468210229346919,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.6324,
659
+ "eval_samples_per_second": 15.872,
660
+ "eval_steps_per_second": 1.985,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5740394659340382,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6500429511070251,
667
+ "learning_rate": 0.00022389213388387174,
668
+ "loss": 0.5283198833465577,
669
+ "mean_token_accuracy": 0.8502798482775689,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5548852207355721,
676
+ "eval_loss": 0.5561797022819519,
677
+ "eval_mean_token_accuracy": 0.8452786498291548,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.5205,
680
+ "eval_samples_per_second": 15.892,
681
+ "eval_steps_per_second": 1.988,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.6020145989954472,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.7056867480278015,
688
+ "learning_rate": 0.00022347395264732053,
689
+ "loss": 0.5400049209594726,
690
+ "mean_token_accuracy": 0.8447613954544068,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5618055154417836,
697
+ "eval_loss": 0.556106686592102,
698
+ "eval_mean_token_accuracy": 0.8465680112672407,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.2971,
701
+ "eval_samples_per_second": 15.933,
702
+ "eval_steps_per_second": 1.993,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5665927153080702,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.5987663865089417,
709
+ "learning_rate": 0.00022302249611363625,
710
+ "loss": 0.5143643856048584,
711
+ "mean_token_accuracy": 0.8529589556157589,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.568248552118623,
718
+ "eval_loss": 0.5476346015930176,
719
+ "eval_mean_token_accuracy": 0.8476775434128073,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.9583,
722
+ "eval_samples_per_second": 15.812,
723
+ "eval_steps_per_second": 1.978,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5673687808215618,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.735261857509613,
730
+ "learning_rate": 0.00022253790043595193,
731
+ "loss": 0.509885597229004,
732
+ "mean_token_accuracy": 0.8537046857178211,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5616967284748721,
739
+ "eval_loss": 0.5439274311065674,
740
+ "eval_mean_token_accuracy": 0.8488946217437123,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.0604,
743
+ "eval_samples_per_second": 15.977,
744
+ "eval_steps_per_second": 1.999,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.5529541682451964,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7014835476875305,
751
+ "learning_rate": 0.00022202031176171442,
752
+ "loss": 0.5078992366790771,
753
+ "mean_token_accuracy": 0.8525233261287213,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5827173320359962,
760
+ "eval_loss": 0.5419450402259827,
761
+ "eval_mean_token_accuracy": 0.8477318609176681,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 85.2984,
764
+ "eval_samples_per_second": 16.12,
765
+ "eval_steps_per_second": 2.016,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5755720350891351,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.705613911151886,
772
+ "learning_rate": 0.00022146988618860824,
773
+ "loss": 0.5181350708007812,
774
+ "mean_token_accuracy": 0.8467609457671642,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5743971356125765,
781
+ "eval_loss": 0.5415896773338318,
782
+ "eval_mean_token_accuracy": 0.847328585940738,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 85.5602,
785
+ "eval_samples_per_second": 16.071,
786
+ "eval_steps_per_second": 2.01,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.561330484598875,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.6722865700721741,
793
+ "learning_rate": 0.0002208867897174789,
794
+ "loss": 0.499837589263916,
795
+ "mean_token_accuracy": 0.8518734864890576,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.5865232653396074,
802
+ "eval_loss": 0.5437926650047302,
803
+ "eval_mean_token_accuracy": 0.8450997017843779,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4116,
806
+ "eval_samples_per_second": 15.912,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.547389242425561,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.7935577034950256,
814
+ "learning_rate": 0.00022027119820226907,
815
+ "loss": 0.4977591514587402,
816
+ "mean_token_accuracy": 0.8539491161704064,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.5290903090391048,
823
+ "eval_loss": 0.5409526824951172,
824
+ "eval_mean_token_accuracy": 0.8497545698354411,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.7262,
827
+ "eval_samples_per_second": 15.854,
828
+ "eval_steps_per_second": 1.983,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5687909748405218,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6180546283721924,
835
+ "learning_rate": 0.00021962329729698345,
836
+ "loss": 0.5109643459320068,
837
+ "mean_token_accuracy": 0.8521598495543004,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.5503541858390321,
844
+ "eval_loss": 0.5361555218696594,
845
+ "eval_mean_token_accuracy": 0.8510884285666221,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.3339,
848
+ "eval_samples_per_second": 15.927,
849
+ "eval_steps_per_second": 1.992,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4739728841261986,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.8058829307556152,
856
+ "learning_rate": 0.0002189432823996982,
857
+ "loss": 0.4204097747802734,
858
+ "mean_token_accuracy": 0.8728981889211215,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.5077334992414297,
865
+ "eval_loss": 0.5531114339828491,
866
+ "eval_mean_token_accuracy": 0.8489257208136625,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.4801,
869
+ "eval_samples_per_second": 15.9,
870
+ "eval_steps_per_second": 1.989,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.4594309840351343,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.6906896829605103,
877
+ "learning_rate": 0.0002182313585936314,
878
+ "loss": 0.4071959495544434,
879
+ "mean_token_accuracy": 0.8732857562601566,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.49850136994622474,
886
+ "eval_loss": 0.5486204624176025,
887
+ "eval_mean_token_accuracy": 0.8507991450470548,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.3364,
890
+ "eval_samples_per_second": 15.926,
891
+ "eval_steps_per_second": 1.992,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4881629109382629,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6343470215797424,
898
+ "learning_rate": 0.0002174877405852928,
899
+ "loss": 0.41669540405273436,
900
+ "mean_token_accuracy": 0.8711295068264008,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.49155513924914734,
907
+ "eval_loss": 0.555109441280365,
908
+ "eval_mean_token_accuracy": 0.8496399400539176,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 86.3295,
911
+ "eval_samples_per_second": 15.927,
912
+ "eval_steps_per_second": 1.992,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.4648668970912695,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.8014165163040161,
919
+ "learning_rate": 0.00021671265263973133,
920
+ "loss": 0.4110250473022461,
921
+ "mean_token_accuracy": 0.8754166305065155,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.4909258722219356,
928
+ "eval_loss": 0.5539511442184448,
929
+ "eval_mean_token_accuracy": 0.8492401502160138,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.3468,
932
+ "eval_samples_per_second": 15.924,
933
+ "eval_steps_per_second": 1.992,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4824485514312983,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6665191054344177,
940
+ "learning_rate": 0.00021590632851289967,
941
+ "loss": 0.4181404113769531,
942
+ "mean_token_accuracy": 0.8726993151009083,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.4986876940657926,
949
+ "eval_loss": 0.547695517539978,
950
+ "eval_mean_token_accuracy": 0.8501384708770486,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.3838,
953
+ "eval_samples_per_second": 15.917,
954
+ "eval_steps_per_second": 1.991,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.4751896943897009,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.81158047914505,
961
+ "learning_rate": 0.00021506901138115678,
962
+ "loss": 0.40689678192138673,
963
+ "mean_token_accuracy": 0.8745221219956875,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.507153491121392,
970
+ "eval_loss": 0.5501641631126404,
971
+ "eval_mean_token_accuracy": 0.8495670116918032,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.0912,
974
+ "eval_samples_per_second": 15.971,
975
+ "eval_steps_per_second": 1.998,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4873133715242147,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7218056321144104,
982
+ "learning_rate": 0.0002142009537679292,
983
+ "loss": 0.42701358795166017,
984
+ "mean_token_accuracy": 0.8695114746689796,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5202612736543943,
991
+ "eval_loss": 0.5491839051246643,
992
+ "eval_mean_token_accuracy": 0.8494071208460386,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.1142,
995
+ "eval_samples_per_second": 15.967,
996
+ "eval_steps_per_second": 1.997,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.4762951169162989,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 0.7194424867630005,
1003
+ "learning_rate": 0.0002133024174675534,
1004
+ "loss": 0.42299847602844237,
1005
+ "mean_token_accuracy": 0.8709790132939815,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4899340462546016,
1012
+ "eval_loss": 0.5522511601448059,
1013
+ "eval_mean_token_accuracy": 0.8492208258357159,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.463,
1016
+ "eval_samples_per_second": 15.903,
1017
+ "eval_steps_per_second": 1.989,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.49650347977876663,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.8406022787094116,
1024
+ "learning_rate": 0.0002123736734663221,
1025
+ "loss": 0.4275330066680908,
1026
+ "mean_token_accuracy": 0.8670595556497573,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.49691385654515996,
1033
+ "eval_loss": 0.5491269826889038,
1034
+ "eval_mean_token_accuracy": 0.850309816210769,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.17,
1037
+ "eval_samples_per_second": 15.957,
1038
+ "eval_steps_per_second": 1.996,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.48843890577554705,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.9082473516464233,
1045
+ "learning_rate": 0.00021141500186075868,
1046
+ "loss": 0.4309722423553467,
1047
+ "mean_token_accuracy": 0.8686766296625137,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5543508351195691,
1054
+ "eval_loss": 0.5478800535202026,
1055
+ "eval_mean_token_accuracy": 0.8478029522784921,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.3835,
1058
+ "eval_samples_per_second": 15.917,
1059
+ "eval_steps_per_second": 1.991,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.4777219031006098,
1064
+ "epoch": 2.5379825653798256,
1065
+ "grad_norm": 0.7448089122772217,
1066
+ "learning_rate": 0.0002104266917731438,
1067
+ "loss": 0.423325252532959,
1068
+ "mean_token_accuracy": 0.8706337086856365,
1069
+ "num_tokens": 2384270.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.5379825653798256,
1074
+ "eval_entropy": 0.49857561550168106,
1075
+ "eval_loss": 0.5511948466300964,
1076
+ "eval_mean_token_accuracy": 0.8502220289651737,
1077
+ "eval_num_tokens": 2384270.0,
1078
+ "eval_runtime": 86.5399,
1079
+ "eval_samples_per_second": 15.889,
1080
+ "eval_steps_per_second": 1.988,
1081
+ "step": 1020
1082
+ }
1083
+ ],
1084
+ "logging_steps": 20,
1085
+ "max_steps": 4020,
1086
+ "num_input_tokens_seen": 0,
1087
+ "num_train_epochs": 10,
1088
+ "save_steps": 20,
1089
+ "stateful_callbacks": {
1090
+ "TrainerControl": {
1091
+ "args": {
1092
+ "should_epoch_stop": false,
1093
+ "should_evaluate": false,
1094
+ "should_log": false,
1095
+ "should_save": true,
1096
+ "should_training_stop": false
1097
+ },
1098
+ "attributes": {}
1099
+ }
1100
+ },
1101
+ "total_flos": 1.0076952699436032e+17,
1102
+ "train_batch_size": 4,
1103
+ "trial_name": null,
1104
+ "trial_params": null
1105
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.028265386974777595,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1040/trainer_state.json ADDED
@@ -0,0 +1,1126 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.587795765877958,
6
+ "eval_steps": 20,
7
+ "global_step": 1040,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.955029806494713,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.020533561706543,
16
+ "learning_rate": 1.0694800202775147e-05,
17
+ "loss": 1.7107986450195312,
18
+ "mean_token_accuracy": 0.6487608112394809,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.3144892034835594,
25
+ "eval_loss": 1.1198534965515137,
26
+ "eval_mean_token_accuracy": 0.7460246955932572,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.0565,
29
+ "eval_samples_per_second": 15.794,
30
+ "eval_steps_per_second": 1.976,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.0063214391469955,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.572906494140625,
37
+ "learning_rate": 2.1952484626748985e-05,
38
+ "loss": 0.8663722991943359,
39
+ "mean_token_accuracy": 0.7779282338917255,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7921617945959402,
46
+ "eval_loss": 0.7062025666236877,
47
+ "eval_mean_token_accuracy": 0.8100443180910376,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.5189,
50
+ "eval_samples_per_second": 15.892,
51
+ "eval_steps_per_second": 1.988,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7682028576731682,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3003711700439453,
58
+ "learning_rate": 3.3210169050722824e-05,
59
+ "loss": 0.673183822631836,
60
+ "mean_token_accuracy": 0.8182129614055157,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7059133584762729,
67
+ "eval_loss": 0.6481946706771851,
68
+ "eval_mean_token_accuracy": 0.8227418761613757,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.5098,
71
+ "eval_samples_per_second": 15.894,
72
+ "eval_steps_per_second": 1.988,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7029960259795189,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.3664201498031616,
79
+ "learning_rate": 4.4467853474696664e-05,
80
+ "loss": 0.6354611873626709,
81
+ "mean_token_accuracy": 0.8243416830897331,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6867308004997498,
88
+ "eval_loss": 0.6179807186126709,
89
+ "eval_mean_token_accuracy": 0.8302594811417335,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.3969,
92
+ "eval_samples_per_second": 15.915,
93
+ "eval_steps_per_second": 1.991,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6764581337571144,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9815880656242371,
100
+ "learning_rate": 5.57255378986705e-05,
101
+ "loss": 0.5988658905029297,
102
+ "mean_token_accuracy": 0.8329168625175953,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6790881479202315,
109
+ "eval_loss": 0.5998476147651672,
110
+ "eval_mean_token_accuracy": 0.8318756420251935,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.6653,
113
+ "eval_samples_per_second": 15.866,
114
+ "eval_steps_per_second": 1.985,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6744543805718421,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.932099461555481,
121
+ "learning_rate": 6.698322232264434e-05,
122
+ "loss": 0.5991750717163086,
123
+ "mean_token_accuracy": 0.8304223112761975,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.6813044282932614,
130
+ "eval_loss": 0.5922021269798279,
131
+ "eval_mean_token_accuracy": 0.8346439617317777,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.1551,
134
+ "eval_samples_per_second": 15.96,
135
+ "eval_steps_per_second": 1.996,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6663189359009266,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9528499841690063,
142
+ "learning_rate": 7.824090674661818e-05,
143
+ "loss": 0.5891091346740722,
144
+ "mean_token_accuracy": 0.832152470946312,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6398407208711602,
151
+ "eval_loss": 0.5859636664390564,
152
+ "eval_mean_token_accuracy": 0.8372074996316156,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.2706,
155
+ "eval_samples_per_second": 15.938,
156
+ "eval_steps_per_second": 1.994,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.64859763905406,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8468204140663147,
163
+ "learning_rate": 8.949859117059201e-05,
164
+ "loss": 0.569426441192627,
165
+ "mean_token_accuracy": 0.8401990942656994,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6381674285891444,
172
+ "eval_loss": 0.5744525790214539,
173
+ "eval_mean_token_accuracy": 0.838626817908398,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.2848,
176
+ "eval_samples_per_second": 15.936,
177
+ "eval_steps_per_second": 1.993,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6432608783245086,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8765804767608643,
184
+ "learning_rate": 0.00010075627559456587,
185
+ "loss": 0.5687318801879883,
186
+ "mean_token_accuracy": 0.839249350130558,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6047098288355872,
193
+ "eval_loss": 0.5679298043251038,
194
+ "eval_mean_token_accuracy": 0.8410577181466791,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.5879,
197
+ "eval_samples_per_second": 15.88,
198
+ "eval_steps_per_second": 1.986,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6440276011824608,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9576020240783691,
205
+ "learning_rate": 0.00011201396001853971,
206
+ "loss": 0.5828506469726562,
207
+ "mean_token_accuracy": 0.837553184479475,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6161119360909906,
214
+ "eval_loss": 0.5702911615371704,
215
+ "eval_mean_token_accuracy": 0.8407089398350827,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3341,
218
+ "eval_samples_per_second": 15.926,
219
+ "eval_steps_per_second": 1.992,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6377195850014686,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7212373614311218,
226
+ "learning_rate": 0.00012327164444251353,
227
+ "loss": 0.5702451229095459,
228
+ "mean_token_accuracy": 0.8397969007492065,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6080108886194784,
235
+ "eval_loss": 0.5633499622344971,
236
+ "eval_mean_token_accuracy": 0.8396634854549585,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.4945,
239
+ "eval_samples_per_second": 15.897,
240
+ "eval_steps_per_second": 1.989,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6287345830351114,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.848779022693634,
247
+ "learning_rate": 0.00013452932886648739,
248
+ "loss": 0.5506546020507812,
249
+ "mean_token_accuracy": 0.8438881888985634,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6307531505130058,
256
+ "eval_loss": 0.5573338270187378,
257
+ "eval_mean_token_accuracy": 0.8431362606758295,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.3535,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.992,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6223786748945713,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.7316951751708984,
268
+ "learning_rate": 0.0001457870132904612,
269
+ "loss": 0.5495625972747803,
270
+ "mean_token_accuracy": 0.8440376669168472,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.623454462476941,
277
+ "eval_loss": 0.5619264245033264,
278
+ "eval_mean_token_accuracy": 0.8431175777385401,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.2008,
281
+ "eval_samples_per_second": 15.951,
282
+ "eval_steps_per_second": 1.995,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6281675305217505,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7639564871788025,
289
+ "learning_rate": 0.00015704469771443506,
290
+ "loss": 0.5604369163513183,
291
+ "mean_token_accuracy": 0.8401600055396556,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.63416675980701,
298
+ "eval_loss": 0.5612760782241821,
299
+ "eval_mean_token_accuracy": 0.842435666294985,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.25,
302
+ "eval_samples_per_second": 15.942,
303
+ "eval_steps_per_second": 1.994,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6427909277379513,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6475813388824463,
310
+ "learning_rate": 0.0001683023821384089,
311
+ "loss": 0.573763370513916,
312
+ "mean_token_accuracy": 0.8370340794324875,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6231539840268534,
319
+ "eval_loss": 0.5566866397857666,
320
+ "eval_mean_token_accuracy": 0.844177934319474,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.4858,
323
+ "eval_samples_per_second": 15.899,
324
+ "eval_steps_per_second": 1.989,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6226776849478484,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.8886699676513672,
331
+ "learning_rate": 0.00017956006656238274,
332
+ "loss": 0.558210802078247,
333
+ "mean_token_accuracy": 0.84083157107234,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6066981683983359,
340
+ "eval_loss": 0.5585207939147949,
341
+ "eval_mean_token_accuracy": 0.8423153311014175,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.3463,
344
+ "eval_samples_per_second": 15.924,
345
+ "eval_steps_per_second": 1.992,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6249004438519478,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8791211843490601,
352
+ "learning_rate": 0.00019081775098635657,
353
+ "loss": 0.5603597164154053,
354
+ "mean_token_accuracy": 0.8420463085174561,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6082247584018596,
361
+ "eval_loss": 0.5616299510002136,
362
+ "eval_mean_token_accuracy": 0.8431286801432454,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.1253,
365
+ "eval_samples_per_second": 15.965,
366
+ "eval_steps_per_second": 1.997,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6362396612763405,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.8606319427490234,
373
+ "learning_rate": 0.0002020754354103304,
374
+ "loss": 0.5735773563385009,
375
+ "mean_token_accuracy": 0.8371490836143494,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6492362072648004,
382
+ "eval_loss": 0.5646467804908752,
383
+ "eval_mean_token_accuracy": 0.8415517574825953,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.3351,
386
+ "eval_samples_per_second": 15.926,
387
+ "eval_steps_per_second": 1.992,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.638665035739541,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.7773950099945068,
394
+ "learning_rate": 0.00021333311983430425,
395
+ "loss": 0.5820859909057617,
396
+ "mean_token_accuracy": 0.8372561208903789,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6434498637221581,
403
+ "eval_loss": 0.5645168423652649,
404
+ "eval_mean_token_accuracy": 0.8420382481674815,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.1216,
407
+ "eval_samples_per_second": 15.966,
408
+ "eval_steps_per_second": 1.997,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6316851265728474,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 1.6120579242706299,
415
+ "learning_rate": 0.00022459080425827807,
416
+ "loss": 0.5637502670288086,
417
+ "mean_token_accuracy": 0.8386227294802666,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6469012776086497,
424
+ "eval_loss": 0.5758090615272522,
425
+ "eval_mean_token_accuracy": 0.8397158470957778,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.6139,
428
+ "eval_samples_per_second": 15.875,
429
+ "eval_steps_per_second": 1.986,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5894816922835815,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 1.1616325378417969,
436
+ "learning_rate": 0.00022626713048053178,
437
+ "loss": 0.5316025257110596,
438
+ "mean_token_accuracy": 0.8466163017810919,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5860798164855602,
445
+ "eval_loss": 0.5777581930160522,
446
+ "eval_mean_token_accuracy": 0.8396938103576039,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.1449,
449
+ "eval_samples_per_second": 15.961,
450
+ "eval_steps_per_second": 1.997,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5818420693278312,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.7999453544616699,
457
+ "learning_rate": 0.00022622107023288778,
458
+ "loss": 0.5221010208129883,
459
+ "mean_token_accuracy": 0.8474301159381866,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5783926014636838,
466
+ "eval_loss": 0.5700300931930542,
467
+ "eval_mean_token_accuracy": 0.8430753537388735,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.5308,
470
+ "eval_samples_per_second": 15.89,
471
+ "eval_steps_per_second": 1.988,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5612493887543678,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 1.015687346458435,
478
+ "learning_rate": 0.00022614090619491568,
479
+ "loss": 0.5084867000579834,
480
+ "mean_token_accuracy": 0.8495561093091964,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5841563874205877,
487
+ "eval_loss": 0.5693665742874146,
488
+ "eval_mean_token_accuracy": 0.8427817298229351,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.5256,
491
+ "eval_samples_per_second": 15.891,
492
+ "eval_steps_per_second": 1.988,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5828216474503278,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 1.9750930070877075,
499
+ "learning_rate": 0.00022602666254299594,
500
+ "loss": 0.5180017948150635,
501
+ "mean_token_accuracy": 0.8515685826539994,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5806607044366903,
508
+ "eval_loss": 0.5804352760314941,
509
+ "eval_mean_token_accuracy": 0.8413014668364858,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.1199,
512
+ "eval_samples_per_second": 15.966,
513
+ "eval_steps_per_second": 1.997,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5926914308220148,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 0.8917353749275208,
520
+ "learning_rate": 0.0002258783737314558,
521
+ "loss": 0.528910779953003,
522
+ "mean_token_accuracy": 0.8486074328422546,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5593361884009006,
529
+ "eval_loss": 0.5675153732299805,
530
+ "eval_mean_token_accuracy": 0.8433507802181466,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7289,
533
+ "eval_samples_per_second": 15.854,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5865630559623242,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7482362985610962,
541
+ "learning_rate": 0.00022569608448217823,
542
+ "loss": 0.5250466823577881,
543
+ "mean_token_accuracy": 0.8477916084229946,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.543057840230853,
550
+ "eval_loss": 0.5671008229255676,
551
+ "eval_mean_token_accuracy": 0.8428726016088973,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.3403,
554
+ "eval_samples_per_second": 15.925,
555
+ "eval_steps_per_second": 1.992,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5870206747204065,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9473814964294434,
562
+ "learning_rate": 0.00022547984977111448,
563
+ "loss": 0.5252370834350586,
564
+ "mean_token_accuracy": 0.8468369916081429,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.590982622878496,
571
+ "eval_loss": 0.5676343441009521,
572
+ "eval_mean_token_accuracy": 0.8429348746011424,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.5168,
575
+ "eval_samples_per_second": 15.893,
576
+ "eval_steps_per_second": 1.988,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5785854265093804,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.9353351593017578,
583
+ "learning_rate": 0.0002252297348117042,
584
+ "loss": 0.5304938316345215,
585
+ "mean_token_accuracy": 0.8463383808732032,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6099918867612995,
592
+ "eval_loss": 0.5620437860488892,
593
+ "eval_mean_token_accuracy": 0.8430728347495545,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.7741,
596
+ "eval_samples_per_second": 15.846,
597
+ "eval_steps_per_second": 1.982,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5768801040947438,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.9198738932609558,
604
+ "learning_rate": 0.0002249458150352077,
605
+ "loss": 0.520513391494751,
606
+ "mean_token_accuracy": 0.8487689301371575,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.6349420670506566,
613
+ "eval_loss": 0.5645340085029602,
614
+ "eval_mean_token_accuracy": 0.8447844597489335,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.3257,
617
+ "eval_samples_per_second": 15.928,
618
+ "eval_steps_per_second": 1.992,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5822233572602272,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.832811176776886,
625
+ "learning_rate": 0.0002246281760679571,
626
+ "loss": 0.5295282363891601,
627
+ "mean_token_accuracy": 0.8504064798355102,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5829724387027496,
634
+ "eval_loss": 0.5612193942070007,
635
+ "eval_mean_token_accuracy": 0.8449643853791925,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6617,
638
+ "eval_samples_per_second": 15.866,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.571855777129531,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.7665547728538513,
646
+ "learning_rate": 0.00022427691370553263,
647
+ "loss": 0.5187931060791016,
648
+ "mean_token_accuracy": 0.8534420043230057,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5623592240519302,
655
+ "eval_loss": 0.5575760006904602,
656
+ "eval_mean_token_accuracy": 0.8468210229346919,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.6324,
659
+ "eval_samples_per_second": 15.872,
660
+ "eval_steps_per_second": 1.985,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5740394659340382,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6500429511070251,
667
+ "learning_rate": 0.00022389213388387174,
668
+ "loss": 0.5283198833465577,
669
+ "mean_token_accuracy": 0.8502798482775689,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5548852207355721,
676
+ "eval_loss": 0.5561797022819519,
677
+ "eval_mean_token_accuracy": 0.8452786498291548,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.5205,
680
+ "eval_samples_per_second": 15.892,
681
+ "eval_steps_per_second": 1.988,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.6020145989954472,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.7056867480278015,
688
+ "learning_rate": 0.00022347395264732053,
689
+ "loss": 0.5400049209594726,
690
+ "mean_token_accuracy": 0.8447613954544068,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5618055154417836,
697
+ "eval_loss": 0.556106686592102,
698
+ "eval_mean_token_accuracy": 0.8465680112672407,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.2971,
701
+ "eval_samples_per_second": 15.933,
702
+ "eval_steps_per_second": 1.993,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5665927153080702,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.5987663865089417,
709
+ "learning_rate": 0.00022302249611363625,
710
+ "loss": 0.5143643856048584,
711
+ "mean_token_accuracy": 0.8529589556157589,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.568248552118623,
718
+ "eval_loss": 0.5476346015930176,
719
+ "eval_mean_token_accuracy": 0.8476775434128073,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.9583,
722
+ "eval_samples_per_second": 15.812,
723
+ "eval_steps_per_second": 1.978,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5673687808215618,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.735261857509613,
730
+ "learning_rate": 0.00022253790043595193,
731
+ "loss": 0.509885597229004,
732
+ "mean_token_accuracy": 0.8537046857178211,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5616967284748721,
739
+ "eval_loss": 0.5439274311065674,
740
+ "eval_mean_token_accuracy": 0.8488946217437123,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.0604,
743
+ "eval_samples_per_second": 15.977,
744
+ "eval_steps_per_second": 1.999,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.5529541682451964,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7014835476875305,
751
+ "learning_rate": 0.00022202031176171442,
752
+ "loss": 0.5078992366790771,
753
+ "mean_token_accuracy": 0.8525233261287213,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5827173320359962,
760
+ "eval_loss": 0.5419450402259827,
761
+ "eval_mean_token_accuracy": 0.8477318609176681,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 85.2984,
764
+ "eval_samples_per_second": 16.12,
765
+ "eval_steps_per_second": 2.016,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5755720350891351,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.705613911151886,
772
+ "learning_rate": 0.00022146988618860824,
773
+ "loss": 0.5181350708007812,
774
+ "mean_token_accuracy": 0.8467609457671642,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5743971356125765,
781
+ "eval_loss": 0.5415896773338318,
782
+ "eval_mean_token_accuracy": 0.847328585940738,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 85.5602,
785
+ "eval_samples_per_second": 16.071,
786
+ "eval_steps_per_second": 2.01,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.561330484598875,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.6722865700721741,
793
+ "learning_rate": 0.0002208867897174789,
794
+ "loss": 0.499837589263916,
795
+ "mean_token_accuracy": 0.8518734864890576,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.5865232653396074,
802
+ "eval_loss": 0.5437926650047302,
803
+ "eval_mean_token_accuracy": 0.8450997017843779,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4116,
806
+ "eval_samples_per_second": 15.912,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.547389242425561,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.7935577034950256,
814
+ "learning_rate": 0.00022027119820226907,
815
+ "loss": 0.4977591514587402,
816
+ "mean_token_accuracy": 0.8539491161704064,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.5290903090391048,
823
+ "eval_loss": 0.5409526824951172,
824
+ "eval_mean_token_accuracy": 0.8497545698354411,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.7262,
827
+ "eval_samples_per_second": 15.854,
828
+ "eval_steps_per_second": 1.983,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5687909748405218,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6180546283721924,
835
+ "learning_rate": 0.00021962329729698345,
836
+ "loss": 0.5109643459320068,
837
+ "mean_token_accuracy": 0.8521598495543004,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.5503541858390321,
844
+ "eval_loss": 0.5361555218696594,
845
+ "eval_mean_token_accuracy": 0.8510884285666221,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.3339,
848
+ "eval_samples_per_second": 15.927,
849
+ "eval_steps_per_second": 1.992,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4739728841261986,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.8058829307556152,
856
+ "learning_rate": 0.0002189432823996982,
857
+ "loss": 0.4204097747802734,
858
+ "mean_token_accuracy": 0.8728981889211215,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.5077334992414297,
865
+ "eval_loss": 0.5531114339828491,
866
+ "eval_mean_token_accuracy": 0.8489257208136625,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.4801,
869
+ "eval_samples_per_second": 15.9,
870
+ "eval_steps_per_second": 1.989,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.4594309840351343,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.6906896829605103,
877
+ "learning_rate": 0.0002182313585936314,
878
+ "loss": 0.4071959495544434,
879
+ "mean_token_accuracy": 0.8732857562601566,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.49850136994622474,
886
+ "eval_loss": 0.5486204624176025,
887
+ "eval_mean_token_accuracy": 0.8507991450470548,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.3364,
890
+ "eval_samples_per_second": 15.926,
891
+ "eval_steps_per_second": 1.992,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4881629109382629,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6343470215797424,
898
+ "learning_rate": 0.0002174877405852928,
899
+ "loss": 0.41669540405273436,
900
+ "mean_token_accuracy": 0.8711295068264008,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.49155513924914734,
907
+ "eval_loss": 0.555109441280365,
908
+ "eval_mean_token_accuracy": 0.8496399400539176,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 86.3295,
911
+ "eval_samples_per_second": 15.927,
912
+ "eval_steps_per_second": 1.992,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.4648668970912695,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.8014165163040161,
919
+ "learning_rate": 0.00021671265263973133,
920
+ "loss": 0.4110250473022461,
921
+ "mean_token_accuracy": 0.8754166305065155,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.4909258722219356,
928
+ "eval_loss": 0.5539511442184448,
929
+ "eval_mean_token_accuracy": 0.8492401502160138,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.3468,
932
+ "eval_samples_per_second": 15.924,
933
+ "eval_steps_per_second": 1.992,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4824485514312983,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6665191054344177,
940
+ "learning_rate": 0.00021590632851289967,
941
+ "loss": 0.4181404113769531,
942
+ "mean_token_accuracy": 0.8726993151009083,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.4986876940657926,
949
+ "eval_loss": 0.547695517539978,
950
+ "eval_mean_token_accuracy": 0.8501384708770486,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.3838,
953
+ "eval_samples_per_second": 15.917,
954
+ "eval_steps_per_second": 1.991,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.4751896943897009,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.81158047914505,
961
+ "learning_rate": 0.00021506901138115678,
962
+ "loss": 0.40689678192138673,
963
+ "mean_token_accuracy": 0.8745221219956875,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.507153491121392,
970
+ "eval_loss": 0.5501641631126404,
971
+ "eval_mean_token_accuracy": 0.8495670116918032,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.0912,
974
+ "eval_samples_per_second": 15.971,
975
+ "eval_steps_per_second": 1.998,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4873133715242147,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7218056321144104,
982
+ "learning_rate": 0.0002142009537679292,
983
+ "loss": 0.42701358795166017,
984
+ "mean_token_accuracy": 0.8695114746689796,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5202612736543943,
991
+ "eval_loss": 0.5491839051246643,
992
+ "eval_mean_token_accuracy": 0.8494071208460386,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.1142,
995
+ "eval_samples_per_second": 15.967,
996
+ "eval_steps_per_second": 1.997,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.4762951169162989,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 0.7194424867630005,
1003
+ "learning_rate": 0.0002133024174675534,
1004
+ "loss": 0.42299847602844237,
1005
+ "mean_token_accuracy": 0.8709790132939815,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4899340462546016,
1012
+ "eval_loss": 0.5522511601448059,
1013
+ "eval_mean_token_accuracy": 0.8492208258357159,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.463,
1016
+ "eval_samples_per_second": 15.903,
1017
+ "eval_steps_per_second": 1.989,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.49650347977876663,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.8406022787094116,
1024
+ "learning_rate": 0.0002123736734663221,
1025
+ "loss": 0.4275330066680908,
1026
+ "mean_token_accuracy": 0.8670595556497573,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.49691385654515996,
1033
+ "eval_loss": 0.5491269826889038,
1034
+ "eval_mean_token_accuracy": 0.850309816210769,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.17,
1037
+ "eval_samples_per_second": 15.957,
1038
+ "eval_steps_per_second": 1.996,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.48843890577554705,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.9082473516464233,
1045
+ "learning_rate": 0.00021141500186075868,
1046
+ "loss": 0.4309722423553467,
1047
+ "mean_token_accuracy": 0.8686766296625137,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5543508351195691,
1054
+ "eval_loss": 0.5478800535202026,
1055
+ "eval_mean_token_accuracy": 0.8478029522784921,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.3835,
1058
+ "eval_samples_per_second": 15.917,
1059
+ "eval_steps_per_second": 1.991,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.4777219031006098,
1064
+ "epoch": 2.5379825653798256,
1065
+ "grad_norm": 0.7448089122772217,
1066
+ "learning_rate": 0.0002104266917731438,
1067
+ "loss": 0.423325252532959,
1068
+ "mean_token_accuracy": 0.8706337086856365,
1069
+ "num_tokens": 2384270.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.5379825653798256,
1074
+ "eval_entropy": 0.49857561550168106,
1075
+ "eval_loss": 0.5511948466300964,
1076
+ "eval_mean_token_accuracy": 0.8502220289651737,
1077
+ "eval_num_tokens": 2384270.0,
1078
+ "eval_runtime": 86.5399,
1079
+ "eval_samples_per_second": 15.889,
1080
+ "eval_steps_per_second": 1.988,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.4844174191355705,
1085
+ "epoch": 2.587795765877958,
1086
+ "grad_norm": 0.794029176235199,
1087
+ "learning_rate": 0.00020940904126432,
1088
+ "loss": 0.4176753044128418,
1089
+ "mean_token_accuracy": 0.873535567522049,
1090
+ "num_tokens": 2428036.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.587795765877958,
1095
+ "eval_entropy": 0.485467542222766,
1096
+ "eval_loss": 0.5539286732673645,
1097
+ "eval_mean_token_accuracy": 0.8495475081510322,
1098
+ "eval_num_tokens": 2428036.0,
1099
+ "eval_runtime": 86.135,
1100
+ "eval_samples_per_second": 15.963,
1101
+ "eval_steps_per_second": 1.997,
1102
+ "step": 1040
1103
+ }
1104
+ ],
1105
+ "logging_steps": 20,
1106
+ "max_steps": 4020,
1107
+ "num_input_tokens_seen": 0,
1108
+ "num_train_epochs": 10,
1109
+ "save_steps": 20,
1110
+ "stateful_callbacks": {
1111
+ "TrainerControl": {
1112
+ "args": {
1113
+ "should_epoch_stop": false,
1114
+ "should_evaluate": false,
1115
+ "should_log": false,
1116
+ "should_save": true,
1117
+ "should_training_stop": false
1118
+ },
1119
+ "attributes": {}
1120
+ }
1121
+ },
1122
+ "total_flos": 1.0254458345271091e+17,
1123
+ "train_batch_size": 4,
1124
+ "trial_name": null,
1125
+ "trial_params": null
1126
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.028265386974777595,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1060/trainer_state.json ADDED
@@ -0,0 +1,1147 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.6376089663760895,
6
+ "eval_steps": 20,
7
+ "global_step": 1060,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.955029806494713,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.020533561706543,
16
+ "learning_rate": 1.0694800202775147e-05,
17
+ "loss": 1.7107986450195312,
18
+ "mean_token_accuracy": 0.6487608112394809,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.3144892034835594,
25
+ "eval_loss": 1.1198534965515137,
26
+ "eval_mean_token_accuracy": 0.7460246955932572,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.0565,
29
+ "eval_samples_per_second": 15.794,
30
+ "eval_steps_per_second": 1.976,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.0063214391469955,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.572906494140625,
37
+ "learning_rate": 2.1952484626748985e-05,
38
+ "loss": 0.8663722991943359,
39
+ "mean_token_accuracy": 0.7779282338917255,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7921617945959402,
46
+ "eval_loss": 0.7062025666236877,
47
+ "eval_mean_token_accuracy": 0.8100443180910376,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.5189,
50
+ "eval_samples_per_second": 15.892,
51
+ "eval_steps_per_second": 1.988,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7682028576731682,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3003711700439453,
58
+ "learning_rate": 3.3210169050722824e-05,
59
+ "loss": 0.673183822631836,
60
+ "mean_token_accuracy": 0.8182129614055157,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7059133584762729,
67
+ "eval_loss": 0.6481946706771851,
68
+ "eval_mean_token_accuracy": 0.8227418761613757,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.5098,
71
+ "eval_samples_per_second": 15.894,
72
+ "eval_steps_per_second": 1.988,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7029960259795189,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.3664201498031616,
79
+ "learning_rate": 4.4467853474696664e-05,
80
+ "loss": 0.6354611873626709,
81
+ "mean_token_accuracy": 0.8243416830897331,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6867308004997498,
88
+ "eval_loss": 0.6179807186126709,
89
+ "eval_mean_token_accuracy": 0.8302594811417335,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.3969,
92
+ "eval_samples_per_second": 15.915,
93
+ "eval_steps_per_second": 1.991,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6764581337571144,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9815880656242371,
100
+ "learning_rate": 5.57255378986705e-05,
101
+ "loss": 0.5988658905029297,
102
+ "mean_token_accuracy": 0.8329168625175953,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6790881479202315,
109
+ "eval_loss": 0.5998476147651672,
110
+ "eval_mean_token_accuracy": 0.8318756420251935,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.6653,
113
+ "eval_samples_per_second": 15.866,
114
+ "eval_steps_per_second": 1.985,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6744543805718421,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.932099461555481,
121
+ "learning_rate": 6.698322232264434e-05,
122
+ "loss": 0.5991750717163086,
123
+ "mean_token_accuracy": 0.8304223112761975,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.6813044282932614,
130
+ "eval_loss": 0.5922021269798279,
131
+ "eval_mean_token_accuracy": 0.8346439617317777,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.1551,
134
+ "eval_samples_per_second": 15.96,
135
+ "eval_steps_per_second": 1.996,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6663189359009266,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9528499841690063,
142
+ "learning_rate": 7.824090674661818e-05,
143
+ "loss": 0.5891091346740722,
144
+ "mean_token_accuracy": 0.832152470946312,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6398407208711602,
151
+ "eval_loss": 0.5859636664390564,
152
+ "eval_mean_token_accuracy": 0.8372074996316156,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.2706,
155
+ "eval_samples_per_second": 15.938,
156
+ "eval_steps_per_second": 1.994,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.64859763905406,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8468204140663147,
163
+ "learning_rate": 8.949859117059201e-05,
164
+ "loss": 0.569426441192627,
165
+ "mean_token_accuracy": 0.8401990942656994,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6381674285891444,
172
+ "eval_loss": 0.5744525790214539,
173
+ "eval_mean_token_accuracy": 0.838626817908398,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.2848,
176
+ "eval_samples_per_second": 15.936,
177
+ "eval_steps_per_second": 1.993,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6432608783245086,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8765804767608643,
184
+ "learning_rate": 0.00010075627559456587,
185
+ "loss": 0.5687318801879883,
186
+ "mean_token_accuracy": 0.839249350130558,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6047098288355872,
193
+ "eval_loss": 0.5679298043251038,
194
+ "eval_mean_token_accuracy": 0.8410577181466791,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.5879,
197
+ "eval_samples_per_second": 15.88,
198
+ "eval_steps_per_second": 1.986,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6440276011824608,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9576020240783691,
205
+ "learning_rate": 0.00011201396001853971,
206
+ "loss": 0.5828506469726562,
207
+ "mean_token_accuracy": 0.837553184479475,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6161119360909906,
214
+ "eval_loss": 0.5702911615371704,
215
+ "eval_mean_token_accuracy": 0.8407089398350827,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3341,
218
+ "eval_samples_per_second": 15.926,
219
+ "eval_steps_per_second": 1.992,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6377195850014686,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7212373614311218,
226
+ "learning_rate": 0.00012327164444251353,
227
+ "loss": 0.5702451229095459,
228
+ "mean_token_accuracy": 0.8397969007492065,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6080108886194784,
235
+ "eval_loss": 0.5633499622344971,
236
+ "eval_mean_token_accuracy": 0.8396634854549585,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.4945,
239
+ "eval_samples_per_second": 15.897,
240
+ "eval_steps_per_second": 1.989,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6287345830351114,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.848779022693634,
247
+ "learning_rate": 0.00013452932886648739,
248
+ "loss": 0.5506546020507812,
249
+ "mean_token_accuracy": 0.8438881888985634,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6307531505130058,
256
+ "eval_loss": 0.5573338270187378,
257
+ "eval_mean_token_accuracy": 0.8431362606758295,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.3535,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.992,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6223786748945713,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.7316951751708984,
268
+ "learning_rate": 0.0001457870132904612,
269
+ "loss": 0.5495625972747803,
270
+ "mean_token_accuracy": 0.8440376669168472,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.623454462476941,
277
+ "eval_loss": 0.5619264245033264,
278
+ "eval_mean_token_accuracy": 0.8431175777385401,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.2008,
281
+ "eval_samples_per_second": 15.951,
282
+ "eval_steps_per_second": 1.995,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6281675305217505,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7639564871788025,
289
+ "learning_rate": 0.00015704469771443506,
290
+ "loss": 0.5604369163513183,
291
+ "mean_token_accuracy": 0.8401600055396556,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.63416675980701,
298
+ "eval_loss": 0.5612760782241821,
299
+ "eval_mean_token_accuracy": 0.842435666294985,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.25,
302
+ "eval_samples_per_second": 15.942,
303
+ "eval_steps_per_second": 1.994,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6427909277379513,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6475813388824463,
310
+ "learning_rate": 0.0001683023821384089,
311
+ "loss": 0.573763370513916,
312
+ "mean_token_accuracy": 0.8370340794324875,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6231539840268534,
319
+ "eval_loss": 0.5566866397857666,
320
+ "eval_mean_token_accuracy": 0.844177934319474,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.4858,
323
+ "eval_samples_per_second": 15.899,
324
+ "eval_steps_per_second": 1.989,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6226776849478484,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.8886699676513672,
331
+ "learning_rate": 0.00017956006656238274,
332
+ "loss": 0.558210802078247,
333
+ "mean_token_accuracy": 0.84083157107234,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6066981683983359,
340
+ "eval_loss": 0.5585207939147949,
341
+ "eval_mean_token_accuracy": 0.8423153311014175,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.3463,
344
+ "eval_samples_per_second": 15.924,
345
+ "eval_steps_per_second": 1.992,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6249004438519478,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8791211843490601,
352
+ "learning_rate": 0.00019081775098635657,
353
+ "loss": 0.5603597164154053,
354
+ "mean_token_accuracy": 0.8420463085174561,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6082247584018596,
361
+ "eval_loss": 0.5616299510002136,
362
+ "eval_mean_token_accuracy": 0.8431286801432454,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.1253,
365
+ "eval_samples_per_second": 15.965,
366
+ "eval_steps_per_second": 1.997,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6362396612763405,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.8606319427490234,
373
+ "learning_rate": 0.0002020754354103304,
374
+ "loss": 0.5735773563385009,
375
+ "mean_token_accuracy": 0.8371490836143494,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6492362072648004,
382
+ "eval_loss": 0.5646467804908752,
383
+ "eval_mean_token_accuracy": 0.8415517574825953,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.3351,
386
+ "eval_samples_per_second": 15.926,
387
+ "eval_steps_per_second": 1.992,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.638665035739541,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.7773950099945068,
394
+ "learning_rate": 0.00021333311983430425,
395
+ "loss": 0.5820859909057617,
396
+ "mean_token_accuracy": 0.8372561208903789,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6434498637221581,
403
+ "eval_loss": 0.5645168423652649,
404
+ "eval_mean_token_accuracy": 0.8420382481674815,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.1216,
407
+ "eval_samples_per_second": 15.966,
408
+ "eval_steps_per_second": 1.997,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6316851265728474,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 1.6120579242706299,
415
+ "learning_rate": 0.00022459080425827807,
416
+ "loss": 0.5637502670288086,
417
+ "mean_token_accuracy": 0.8386227294802666,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6469012776086497,
424
+ "eval_loss": 0.5758090615272522,
425
+ "eval_mean_token_accuracy": 0.8397158470957778,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.6139,
428
+ "eval_samples_per_second": 15.875,
429
+ "eval_steps_per_second": 1.986,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5894816922835815,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 1.1616325378417969,
436
+ "learning_rate": 0.00022626713048053178,
437
+ "loss": 0.5316025257110596,
438
+ "mean_token_accuracy": 0.8466163017810919,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5860798164855602,
445
+ "eval_loss": 0.5777581930160522,
446
+ "eval_mean_token_accuracy": 0.8396938103576039,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.1449,
449
+ "eval_samples_per_second": 15.961,
450
+ "eval_steps_per_second": 1.997,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5818420693278312,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.7999453544616699,
457
+ "learning_rate": 0.00022622107023288778,
458
+ "loss": 0.5221010208129883,
459
+ "mean_token_accuracy": 0.8474301159381866,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5783926014636838,
466
+ "eval_loss": 0.5700300931930542,
467
+ "eval_mean_token_accuracy": 0.8430753537388735,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.5308,
470
+ "eval_samples_per_second": 15.89,
471
+ "eval_steps_per_second": 1.988,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5612493887543678,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 1.015687346458435,
478
+ "learning_rate": 0.00022614090619491568,
479
+ "loss": 0.5084867000579834,
480
+ "mean_token_accuracy": 0.8495561093091964,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5841563874205877,
487
+ "eval_loss": 0.5693665742874146,
488
+ "eval_mean_token_accuracy": 0.8427817298229351,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.5256,
491
+ "eval_samples_per_second": 15.891,
492
+ "eval_steps_per_second": 1.988,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5828216474503278,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 1.9750930070877075,
499
+ "learning_rate": 0.00022602666254299594,
500
+ "loss": 0.5180017948150635,
501
+ "mean_token_accuracy": 0.8515685826539994,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5806607044366903,
508
+ "eval_loss": 0.5804352760314941,
509
+ "eval_mean_token_accuracy": 0.8413014668364858,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.1199,
512
+ "eval_samples_per_second": 15.966,
513
+ "eval_steps_per_second": 1.997,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5926914308220148,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 0.8917353749275208,
520
+ "learning_rate": 0.0002258783737314558,
521
+ "loss": 0.528910779953003,
522
+ "mean_token_accuracy": 0.8486074328422546,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5593361884009006,
529
+ "eval_loss": 0.5675153732299805,
530
+ "eval_mean_token_accuracy": 0.8433507802181466,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7289,
533
+ "eval_samples_per_second": 15.854,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5865630559623242,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7482362985610962,
541
+ "learning_rate": 0.00022569608448217823,
542
+ "loss": 0.5250466823577881,
543
+ "mean_token_accuracy": 0.8477916084229946,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.543057840230853,
550
+ "eval_loss": 0.5671008229255676,
551
+ "eval_mean_token_accuracy": 0.8428726016088973,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.3403,
554
+ "eval_samples_per_second": 15.925,
555
+ "eval_steps_per_second": 1.992,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5870206747204065,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9473814964294434,
562
+ "learning_rate": 0.00022547984977111448,
563
+ "loss": 0.5252370834350586,
564
+ "mean_token_accuracy": 0.8468369916081429,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.590982622878496,
571
+ "eval_loss": 0.5676343441009521,
572
+ "eval_mean_token_accuracy": 0.8429348746011424,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.5168,
575
+ "eval_samples_per_second": 15.893,
576
+ "eval_steps_per_second": 1.988,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5785854265093804,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.9353351593017578,
583
+ "learning_rate": 0.0002252297348117042,
584
+ "loss": 0.5304938316345215,
585
+ "mean_token_accuracy": 0.8463383808732032,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6099918867612995,
592
+ "eval_loss": 0.5620437860488892,
593
+ "eval_mean_token_accuracy": 0.8430728347495545,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.7741,
596
+ "eval_samples_per_second": 15.846,
597
+ "eval_steps_per_second": 1.982,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5768801040947438,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.9198738932609558,
604
+ "learning_rate": 0.0002249458150352077,
605
+ "loss": 0.520513391494751,
606
+ "mean_token_accuracy": 0.8487689301371575,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.6349420670506566,
613
+ "eval_loss": 0.5645340085029602,
614
+ "eval_mean_token_accuracy": 0.8447844597489335,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.3257,
617
+ "eval_samples_per_second": 15.928,
618
+ "eval_steps_per_second": 1.992,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5822233572602272,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.832811176776886,
625
+ "learning_rate": 0.0002246281760679571,
626
+ "loss": 0.5295282363891601,
627
+ "mean_token_accuracy": 0.8504064798355102,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5829724387027496,
634
+ "eval_loss": 0.5612193942070007,
635
+ "eval_mean_token_accuracy": 0.8449643853791925,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6617,
638
+ "eval_samples_per_second": 15.866,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.571855777129531,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.7665547728538513,
646
+ "learning_rate": 0.00022427691370553263,
647
+ "loss": 0.5187931060791016,
648
+ "mean_token_accuracy": 0.8534420043230057,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5623592240519302,
655
+ "eval_loss": 0.5575760006904602,
656
+ "eval_mean_token_accuracy": 0.8468210229346919,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.6324,
659
+ "eval_samples_per_second": 15.872,
660
+ "eval_steps_per_second": 1.985,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5740394659340382,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6500429511070251,
667
+ "learning_rate": 0.00022389213388387174,
668
+ "loss": 0.5283198833465577,
669
+ "mean_token_accuracy": 0.8502798482775689,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5548852207355721,
676
+ "eval_loss": 0.5561797022819519,
677
+ "eval_mean_token_accuracy": 0.8452786498291548,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.5205,
680
+ "eval_samples_per_second": 15.892,
681
+ "eval_steps_per_second": 1.988,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.6020145989954472,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.7056867480278015,
688
+ "learning_rate": 0.00022347395264732053,
689
+ "loss": 0.5400049209594726,
690
+ "mean_token_accuracy": 0.8447613954544068,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5618055154417836,
697
+ "eval_loss": 0.556106686592102,
698
+ "eval_mean_token_accuracy": 0.8465680112672407,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.2971,
701
+ "eval_samples_per_second": 15.933,
702
+ "eval_steps_per_second": 1.993,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5665927153080702,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.5987663865089417,
709
+ "learning_rate": 0.00022302249611363625,
710
+ "loss": 0.5143643856048584,
711
+ "mean_token_accuracy": 0.8529589556157589,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.568248552118623,
718
+ "eval_loss": 0.5476346015930176,
719
+ "eval_mean_token_accuracy": 0.8476775434128073,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.9583,
722
+ "eval_samples_per_second": 15.812,
723
+ "eval_steps_per_second": 1.978,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5673687808215618,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.735261857509613,
730
+ "learning_rate": 0.00022253790043595193,
731
+ "loss": 0.509885597229004,
732
+ "mean_token_accuracy": 0.8537046857178211,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5616967284748721,
739
+ "eval_loss": 0.5439274311065674,
740
+ "eval_mean_token_accuracy": 0.8488946217437123,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.0604,
743
+ "eval_samples_per_second": 15.977,
744
+ "eval_steps_per_second": 1.999,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.5529541682451964,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7014835476875305,
751
+ "learning_rate": 0.00022202031176171442,
752
+ "loss": 0.5078992366790771,
753
+ "mean_token_accuracy": 0.8525233261287213,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5827173320359962,
760
+ "eval_loss": 0.5419450402259827,
761
+ "eval_mean_token_accuracy": 0.8477318609176681,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 85.2984,
764
+ "eval_samples_per_second": 16.12,
765
+ "eval_steps_per_second": 2.016,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5755720350891351,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.705613911151886,
772
+ "learning_rate": 0.00022146988618860824,
773
+ "loss": 0.5181350708007812,
774
+ "mean_token_accuracy": 0.8467609457671642,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5743971356125765,
781
+ "eval_loss": 0.5415896773338318,
782
+ "eval_mean_token_accuracy": 0.847328585940738,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 85.5602,
785
+ "eval_samples_per_second": 16.071,
786
+ "eval_steps_per_second": 2.01,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.561330484598875,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.6722865700721741,
793
+ "learning_rate": 0.0002208867897174789,
794
+ "loss": 0.499837589263916,
795
+ "mean_token_accuracy": 0.8518734864890576,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.5865232653396074,
802
+ "eval_loss": 0.5437926650047302,
803
+ "eval_mean_token_accuracy": 0.8450997017843779,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4116,
806
+ "eval_samples_per_second": 15.912,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.547389242425561,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.7935577034950256,
814
+ "learning_rate": 0.00022027119820226907,
815
+ "loss": 0.4977591514587402,
816
+ "mean_token_accuracy": 0.8539491161704064,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.5290903090391048,
823
+ "eval_loss": 0.5409526824951172,
824
+ "eval_mean_token_accuracy": 0.8497545698354411,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.7262,
827
+ "eval_samples_per_second": 15.854,
828
+ "eval_steps_per_second": 1.983,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5687909748405218,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6180546283721924,
835
+ "learning_rate": 0.00021962329729698345,
836
+ "loss": 0.5109643459320068,
837
+ "mean_token_accuracy": 0.8521598495543004,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.5503541858390321,
844
+ "eval_loss": 0.5361555218696594,
845
+ "eval_mean_token_accuracy": 0.8510884285666221,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.3339,
848
+ "eval_samples_per_second": 15.927,
849
+ "eval_steps_per_second": 1.992,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4739728841261986,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.8058829307556152,
856
+ "learning_rate": 0.0002189432823996982,
857
+ "loss": 0.4204097747802734,
858
+ "mean_token_accuracy": 0.8728981889211215,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.5077334992414297,
865
+ "eval_loss": 0.5531114339828491,
866
+ "eval_mean_token_accuracy": 0.8489257208136625,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.4801,
869
+ "eval_samples_per_second": 15.9,
870
+ "eval_steps_per_second": 1.989,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.4594309840351343,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.6906896829605103,
877
+ "learning_rate": 0.0002182313585936314,
878
+ "loss": 0.4071959495544434,
879
+ "mean_token_accuracy": 0.8732857562601566,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.49850136994622474,
886
+ "eval_loss": 0.5486204624176025,
887
+ "eval_mean_token_accuracy": 0.8507991450470548,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.3364,
890
+ "eval_samples_per_second": 15.926,
891
+ "eval_steps_per_second": 1.992,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4881629109382629,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6343470215797424,
898
+ "learning_rate": 0.0002174877405852928,
899
+ "loss": 0.41669540405273436,
900
+ "mean_token_accuracy": 0.8711295068264008,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.49155513924914734,
907
+ "eval_loss": 0.555109441280365,
908
+ "eval_mean_token_accuracy": 0.8496399400539176,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 86.3295,
911
+ "eval_samples_per_second": 15.927,
912
+ "eval_steps_per_second": 1.992,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.4648668970912695,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.8014165163040161,
919
+ "learning_rate": 0.00021671265263973133,
920
+ "loss": 0.4110250473022461,
921
+ "mean_token_accuracy": 0.8754166305065155,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.4909258722219356,
928
+ "eval_loss": 0.5539511442184448,
929
+ "eval_mean_token_accuracy": 0.8492401502160138,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.3468,
932
+ "eval_samples_per_second": 15.924,
933
+ "eval_steps_per_second": 1.992,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4824485514312983,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6665191054344177,
940
+ "learning_rate": 0.00021590632851289967,
941
+ "loss": 0.4181404113769531,
942
+ "mean_token_accuracy": 0.8726993151009083,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.4986876940657926,
949
+ "eval_loss": 0.547695517539978,
950
+ "eval_mean_token_accuracy": 0.8501384708770486,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.3838,
953
+ "eval_samples_per_second": 15.917,
954
+ "eval_steps_per_second": 1.991,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.4751896943897009,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.81158047914505,
961
+ "learning_rate": 0.00021506901138115678,
962
+ "loss": 0.40689678192138673,
963
+ "mean_token_accuracy": 0.8745221219956875,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.507153491121392,
970
+ "eval_loss": 0.5501641631126404,
971
+ "eval_mean_token_accuracy": 0.8495670116918032,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.0912,
974
+ "eval_samples_per_second": 15.971,
975
+ "eval_steps_per_second": 1.998,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4873133715242147,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7218056321144104,
982
+ "learning_rate": 0.0002142009537679292,
983
+ "loss": 0.42701358795166017,
984
+ "mean_token_accuracy": 0.8695114746689796,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5202612736543943,
991
+ "eval_loss": 0.5491839051246643,
992
+ "eval_mean_token_accuracy": 0.8494071208460386,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.1142,
995
+ "eval_samples_per_second": 15.967,
996
+ "eval_steps_per_second": 1.997,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.4762951169162989,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 0.7194424867630005,
1003
+ "learning_rate": 0.0002133024174675534,
1004
+ "loss": 0.42299847602844237,
1005
+ "mean_token_accuracy": 0.8709790132939815,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4899340462546016,
1012
+ "eval_loss": 0.5522511601448059,
1013
+ "eval_mean_token_accuracy": 0.8492208258357159,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.463,
1016
+ "eval_samples_per_second": 15.903,
1017
+ "eval_steps_per_second": 1.989,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.49650347977876663,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.8406022787094116,
1024
+ "learning_rate": 0.0002123736734663221,
1025
+ "loss": 0.4275330066680908,
1026
+ "mean_token_accuracy": 0.8670595556497573,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.49691385654515996,
1033
+ "eval_loss": 0.5491269826889038,
1034
+ "eval_mean_token_accuracy": 0.850309816210769,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.17,
1037
+ "eval_samples_per_second": 15.957,
1038
+ "eval_steps_per_second": 1.996,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.48843890577554705,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.9082473516464233,
1045
+ "learning_rate": 0.00021141500186075868,
1046
+ "loss": 0.4309722423553467,
1047
+ "mean_token_accuracy": 0.8686766296625137,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5543508351195691,
1054
+ "eval_loss": 0.5478800535202026,
1055
+ "eval_mean_token_accuracy": 0.8478029522784921,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.3835,
1058
+ "eval_samples_per_second": 15.917,
1059
+ "eval_steps_per_second": 1.991,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.4777219031006098,
1064
+ "epoch": 2.5379825653798256,
1065
+ "grad_norm": 0.7448089122772217,
1066
+ "learning_rate": 0.0002104266917731438,
1067
+ "loss": 0.423325252532959,
1068
+ "mean_token_accuracy": 0.8706337086856365,
1069
+ "num_tokens": 2384270.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.5379825653798256,
1074
+ "eval_entropy": 0.49857561550168106,
1075
+ "eval_loss": 0.5511948466300964,
1076
+ "eval_mean_token_accuracy": 0.8502220289651737,
1077
+ "eval_num_tokens": 2384270.0,
1078
+ "eval_runtime": 86.5399,
1079
+ "eval_samples_per_second": 15.889,
1080
+ "eval_steps_per_second": 1.988,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.4844174191355705,
1085
+ "epoch": 2.587795765877958,
1086
+ "grad_norm": 0.794029176235199,
1087
+ "learning_rate": 0.00020940904126432,
1088
+ "loss": 0.4176753044128418,
1089
+ "mean_token_accuracy": 0.873535567522049,
1090
+ "num_tokens": 2428036.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.587795765877958,
1095
+ "eval_entropy": 0.485467542222766,
1096
+ "eval_loss": 0.5539286732673645,
1097
+ "eval_mean_token_accuracy": 0.8495475081510322,
1098
+ "eval_num_tokens": 2428036.0,
1099
+ "eval_runtime": 86.135,
1100
+ "eval_samples_per_second": 15.963,
1101
+ "eval_steps_per_second": 1.997,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.49070929251611234,
1106
+ "epoch": 2.6376089663760895,
1107
+ "grad_norm": 0.7558256983757019,
1108
+ "learning_rate": 0.0002083623572438007,
1109
+ "loss": 0.42867293357849123,
1110
+ "mean_token_accuracy": 0.8696666076779366,
1111
+ "num_tokens": 2476815.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.6376089663760895,
1116
+ "eval_entropy": 0.490822730889154,
1117
+ "eval_loss": 0.5434785485267639,
1118
+ "eval_mean_token_accuracy": 0.850568296950917,
1119
+ "eval_num_tokens": 2476815.0,
1120
+ "eval_runtime": 86.4933,
1121
+ "eval_samples_per_second": 15.897,
1122
+ "eval_steps_per_second": 1.989,
1123
+ "step": 1060
1124
+ }
1125
+ ],
1126
+ "logging_steps": 20,
1127
+ "max_steps": 4020,
1128
+ "num_input_tokens_seen": 0,
1129
+ "num_train_epochs": 10,
1130
+ "save_steps": 20,
1131
+ "stateful_callbacks": {
1132
+ "TrainerControl": {
1133
+ "args": {
1134
+ "should_epoch_stop": false,
1135
+ "should_evaluate": false,
1136
+ "should_log": false,
1137
+ "should_save": true,
1138
+ "should_training_stop": false
1139
+ },
1140
+ "attributes": {}
1141
+ }
1142
+ },
1143
+ "total_flos": 1.045743734380032e+17,
1144
+ "train_batch_size": 4,
1145
+ "trial_name": null,
1146
+ "trial_params": null
1147
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.028265386974777595,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1080/trainer_state.json ADDED
@@ -0,0 +1,1168 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.6874221668742218,
6
+ "eval_steps": 20,
7
+ "global_step": 1080,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.955029806494713,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.020533561706543,
16
+ "learning_rate": 1.0694800202775147e-05,
17
+ "loss": 1.7107986450195312,
18
+ "mean_token_accuracy": 0.6487608112394809,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.3144892034835594,
25
+ "eval_loss": 1.1198534965515137,
26
+ "eval_mean_token_accuracy": 0.7460246955932572,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.0565,
29
+ "eval_samples_per_second": 15.794,
30
+ "eval_steps_per_second": 1.976,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.0063214391469955,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.572906494140625,
37
+ "learning_rate": 2.1952484626748985e-05,
38
+ "loss": 0.8663722991943359,
39
+ "mean_token_accuracy": 0.7779282338917255,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7921617945959402,
46
+ "eval_loss": 0.7062025666236877,
47
+ "eval_mean_token_accuracy": 0.8100443180910376,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.5189,
50
+ "eval_samples_per_second": 15.892,
51
+ "eval_steps_per_second": 1.988,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7682028576731682,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3003711700439453,
58
+ "learning_rate": 3.3210169050722824e-05,
59
+ "loss": 0.673183822631836,
60
+ "mean_token_accuracy": 0.8182129614055157,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7059133584762729,
67
+ "eval_loss": 0.6481946706771851,
68
+ "eval_mean_token_accuracy": 0.8227418761613757,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.5098,
71
+ "eval_samples_per_second": 15.894,
72
+ "eval_steps_per_second": 1.988,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7029960259795189,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.3664201498031616,
79
+ "learning_rate": 4.4467853474696664e-05,
80
+ "loss": 0.6354611873626709,
81
+ "mean_token_accuracy": 0.8243416830897331,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6867308004997498,
88
+ "eval_loss": 0.6179807186126709,
89
+ "eval_mean_token_accuracy": 0.8302594811417335,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.3969,
92
+ "eval_samples_per_second": 15.915,
93
+ "eval_steps_per_second": 1.991,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6764581337571144,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9815880656242371,
100
+ "learning_rate": 5.57255378986705e-05,
101
+ "loss": 0.5988658905029297,
102
+ "mean_token_accuracy": 0.8329168625175953,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6790881479202315,
109
+ "eval_loss": 0.5998476147651672,
110
+ "eval_mean_token_accuracy": 0.8318756420251935,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.6653,
113
+ "eval_samples_per_second": 15.866,
114
+ "eval_steps_per_second": 1.985,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6744543805718421,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.932099461555481,
121
+ "learning_rate": 6.698322232264434e-05,
122
+ "loss": 0.5991750717163086,
123
+ "mean_token_accuracy": 0.8304223112761975,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.6813044282932614,
130
+ "eval_loss": 0.5922021269798279,
131
+ "eval_mean_token_accuracy": 0.8346439617317777,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.1551,
134
+ "eval_samples_per_second": 15.96,
135
+ "eval_steps_per_second": 1.996,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6663189359009266,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9528499841690063,
142
+ "learning_rate": 7.824090674661818e-05,
143
+ "loss": 0.5891091346740722,
144
+ "mean_token_accuracy": 0.832152470946312,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6398407208711602,
151
+ "eval_loss": 0.5859636664390564,
152
+ "eval_mean_token_accuracy": 0.8372074996316156,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.2706,
155
+ "eval_samples_per_second": 15.938,
156
+ "eval_steps_per_second": 1.994,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.64859763905406,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8468204140663147,
163
+ "learning_rate": 8.949859117059201e-05,
164
+ "loss": 0.569426441192627,
165
+ "mean_token_accuracy": 0.8401990942656994,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6381674285891444,
172
+ "eval_loss": 0.5744525790214539,
173
+ "eval_mean_token_accuracy": 0.838626817908398,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.2848,
176
+ "eval_samples_per_second": 15.936,
177
+ "eval_steps_per_second": 1.993,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6432608783245086,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8765804767608643,
184
+ "learning_rate": 0.00010075627559456587,
185
+ "loss": 0.5687318801879883,
186
+ "mean_token_accuracy": 0.839249350130558,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6047098288355872,
193
+ "eval_loss": 0.5679298043251038,
194
+ "eval_mean_token_accuracy": 0.8410577181466791,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.5879,
197
+ "eval_samples_per_second": 15.88,
198
+ "eval_steps_per_second": 1.986,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6440276011824608,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9576020240783691,
205
+ "learning_rate": 0.00011201396001853971,
206
+ "loss": 0.5828506469726562,
207
+ "mean_token_accuracy": 0.837553184479475,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6161119360909906,
214
+ "eval_loss": 0.5702911615371704,
215
+ "eval_mean_token_accuracy": 0.8407089398350827,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3341,
218
+ "eval_samples_per_second": 15.926,
219
+ "eval_steps_per_second": 1.992,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6377195850014686,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7212373614311218,
226
+ "learning_rate": 0.00012327164444251353,
227
+ "loss": 0.5702451229095459,
228
+ "mean_token_accuracy": 0.8397969007492065,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6080108886194784,
235
+ "eval_loss": 0.5633499622344971,
236
+ "eval_mean_token_accuracy": 0.8396634854549585,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.4945,
239
+ "eval_samples_per_second": 15.897,
240
+ "eval_steps_per_second": 1.989,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6287345830351114,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.848779022693634,
247
+ "learning_rate": 0.00013452932886648739,
248
+ "loss": 0.5506546020507812,
249
+ "mean_token_accuracy": 0.8438881888985634,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6307531505130058,
256
+ "eval_loss": 0.5573338270187378,
257
+ "eval_mean_token_accuracy": 0.8431362606758295,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.3535,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.992,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6223786748945713,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.7316951751708984,
268
+ "learning_rate": 0.0001457870132904612,
269
+ "loss": 0.5495625972747803,
270
+ "mean_token_accuracy": 0.8440376669168472,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.623454462476941,
277
+ "eval_loss": 0.5619264245033264,
278
+ "eval_mean_token_accuracy": 0.8431175777385401,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.2008,
281
+ "eval_samples_per_second": 15.951,
282
+ "eval_steps_per_second": 1.995,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6281675305217505,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7639564871788025,
289
+ "learning_rate": 0.00015704469771443506,
290
+ "loss": 0.5604369163513183,
291
+ "mean_token_accuracy": 0.8401600055396556,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.63416675980701,
298
+ "eval_loss": 0.5612760782241821,
299
+ "eval_mean_token_accuracy": 0.842435666294985,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.25,
302
+ "eval_samples_per_second": 15.942,
303
+ "eval_steps_per_second": 1.994,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6427909277379513,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6475813388824463,
310
+ "learning_rate": 0.0001683023821384089,
311
+ "loss": 0.573763370513916,
312
+ "mean_token_accuracy": 0.8370340794324875,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6231539840268534,
319
+ "eval_loss": 0.5566866397857666,
320
+ "eval_mean_token_accuracy": 0.844177934319474,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.4858,
323
+ "eval_samples_per_second": 15.899,
324
+ "eval_steps_per_second": 1.989,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6226776849478484,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.8886699676513672,
331
+ "learning_rate": 0.00017956006656238274,
332
+ "loss": 0.558210802078247,
333
+ "mean_token_accuracy": 0.84083157107234,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6066981683983359,
340
+ "eval_loss": 0.5585207939147949,
341
+ "eval_mean_token_accuracy": 0.8423153311014175,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.3463,
344
+ "eval_samples_per_second": 15.924,
345
+ "eval_steps_per_second": 1.992,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6249004438519478,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8791211843490601,
352
+ "learning_rate": 0.00019081775098635657,
353
+ "loss": 0.5603597164154053,
354
+ "mean_token_accuracy": 0.8420463085174561,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6082247584018596,
361
+ "eval_loss": 0.5616299510002136,
362
+ "eval_mean_token_accuracy": 0.8431286801432454,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.1253,
365
+ "eval_samples_per_second": 15.965,
366
+ "eval_steps_per_second": 1.997,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6362396612763405,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.8606319427490234,
373
+ "learning_rate": 0.0002020754354103304,
374
+ "loss": 0.5735773563385009,
375
+ "mean_token_accuracy": 0.8371490836143494,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6492362072648004,
382
+ "eval_loss": 0.5646467804908752,
383
+ "eval_mean_token_accuracy": 0.8415517574825953,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.3351,
386
+ "eval_samples_per_second": 15.926,
387
+ "eval_steps_per_second": 1.992,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.638665035739541,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.7773950099945068,
394
+ "learning_rate": 0.00021333311983430425,
395
+ "loss": 0.5820859909057617,
396
+ "mean_token_accuracy": 0.8372561208903789,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6434498637221581,
403
+ "eval_loss": 0.5645168423652649,
404
+ "eval_mean_token_accuracy": 0.8420382481674815,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.1216,
407
+ "eval_samples_per_second": 15.966,
408
+ "eval_steps_per_second": 1.997,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6316851265728474,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 1.6120579242706299,
415
+ "learning_rate": 0.00022459080425827807,
416
+ "loss": 0.5637502670288086,
417
+ "mean_token_accuracy": 0.8386227294802666,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6469012776086497,
424
+ "eval_loss": 0.5758090615272522,
425
+ "eval_mean_token_accuracy": 0.8397158470957778,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.6139,
428
+ "eval_samples_per_second": 15.875,
429
+ "eval_steps_per_second": 1.986,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5894816922835815,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 1.1616325378417969,
436
+ "learning_rate": 0.00022626713048053178,
437
+ "loss": 0.5316025257110596,
438
+ "mean_token_accuracy": 0.8466163017810919,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5860798164855602,
445
+ "eval_loss": 0.5777581930160522,
446
+ "eval_mean_token_accuracy": 0.8396938103576039,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.1449,
449
+ "eval_samples_per_second": 15.961,
450
+ "eval_steps_per_second": 1.997,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5818420693278312,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.7999453544616699,
457
+ "learning_rate": 0.00022622107023288778,
458
+ "loss": 0.5221010208129883,
459
+ "mean_token_accuracy": 0.8474301159381866,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5783926014636838,
466
+ "eval_loss": 0.5700300931930542,
467
+ "eval_mean_token_accuracy": 0.8430753537388735,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.5308,
470
+ "eval_samples_per_second": 15.89,
471
+ "eval_steps_per_second": 1.988,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5612493887543678,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 1.015687346458435,
478
+ "learning_rate": 0.00022614090619491568,
479
+ "loss": 0.5084867000579834,
480
+ "mean_token_accuracy": 0.8495561093091964,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5841563874205877,
487
+ "eval_loss": 0.5693665742874146,
488
+ "eval_mean_token_accuracy": 0.8427817298229351,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.5256,
491
+ "eval_samples_per_second": 15.891,
492
+ "eval_steps_per_second": 1.988,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5828216474503278,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 1.9750930070877075,
499
+ "learning_rate": 0.00022602666254299594,
500
+ "loss": 0.5180017948150635,
501
+ "mean_token_accuracy": 0.8515685826539994,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5806607044366903,
508
+ "eval_loss": 0.5804352760314941,
509
+ "eval_mean_token_accuracy": 0.8413014668364858,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.1199,
512
+ "eval_samples_per_second": 15.966,
513
+ "eval_steps_per_second": 1.997,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5926914308220148,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 0.8917353749275208,
520
+ "learning_rate": 0.0002258783737314558,
521
+ "loss": 0.528910779953003,
522
+ "mean_token_accuracy": 0.8486074328422546,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5593361884009006,
529
+ "eval_loss": 0.5675153732299805,
530
+ "eval_mean_token_accuracy": 0.8433507802181466,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7289,
533
+ "eval_samples_per_second": 15.854,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5865630559623242,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7482362985610962,
541
+ "learning_rate": 0.00022569608448217823,
542
+ "loss": 0.5250466823577881,
543
+ "mean_token_accuracy": 0.8477916084229946,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.543057840230853,
550
+ "eval_loss": 0.5671008229255676,
551
+ "eval_mean_token_accuracy": 0.8428726016088973,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.3403,
554
+ "eval_samples_per_second": 15.925,
555
+ "eval_steps_per_second": 1.992,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5870206747204065,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9473814964294434,
562
+ "learning_rate": 0.00022547984977111448,
563
+ "loss": 0.5252370834350586,
564
+ "mean_token_accuracy": 0.8468369916081429,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.590982622878496,
571
+ "eval_loss": 0.5676343441009521,
572
+ "eval_mean_token_accuracy": 0.8429348746011424,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.5168,
575
+ "eval_samples_per_second": 15.893,
576
+ "eval_steps_per_second": 1.988,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5785854265093804,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.9353351593017578,
583
+ "learning_rate": 0.0002252297348117042,
584
+ "loss": 0.5304938316345215,
585
+ "mean_token_accuracy": 0.8463383808732032,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6099918867612995,
592
+ "eval_loss": 0.5620437860488892,
593
+ "eval_mean_token_accuracy": 0.8430728347495545,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.7741,
596
+ "eval_samples_per_second": 15.846,
597
+ "eval_steps_per_second": 1.982,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5768801040947438,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.9198738932609558,
604
+ "learning_rate": 0.0002249458150352077,
605
+ "loss": 0.520513391494751,
606
+ "mean_token_accuracy": 0.8487689301371575,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.6349420670506566,
613
+ "eval_loss": 0.5645340085029602,
614
+ "eval_mean_token_accuracy": 0.8447844597489335,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.3257,
617
+ "eval_samples_per_second": 15.928,
618
+ "eval_steps_per_second": 1.992,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5822233572602272,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.832811176776886,
625
+ "learning_rate": 0.0002246281760679571,
626
+ "loss": 0.5295282363891601,
627
+ "mean_token_accuracy": 0.8504064798355102,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5829724387027496,
634
+ "eval_loss": 0.5612193942070007,
635
+ "eval_mean_token_accuracy": 0.8449643853791925,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6617,
638
+ "eval_samples_per_second": 15.866,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.571855777129531,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.7665547728538513,
646
+ "learning_rate": 0.00022427691370553263,
647
+ "loss": 0.5187931060791016,
648
+ "mean_token_accuracy": 0.8534420043230057,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5623592240519302,
655
+ "eval_loss": 0.5575760006904602,
656
+ "eval_mean_token_accuracy": 0.8468210229346919,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.6324,
659
+ "eval_samples_per_second": 15.872,
660
+ "eval_steps_per_second": 1.985,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5740394659340382,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6500429511070251,
667
+ "learning_rate": 0.00022389213388387174,
668
+ "loss": 0.5283198833465577,
669
+ "mean_token_accuracy": 0.8502798482775689,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5548852207355721,
676
+ "eval_loss": 0.5561797022819519,
677
+ "eval_mean_token_accuracy": 0.8452786498291548,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.5205,
680
+ "eval_samples_per_second": 15.892,
681
+ "eval_steps_per_second": 1.988,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.6020145989954472,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.7056867480278015,
688
+ "learning_rate": 0.00022347395264732053,
689
+ "loss": 0.5400049209594726,
690
+ "mean_token_accuracy": 0.8447613954544068,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5618055154417836,
697
+ "eval_loss": 0.556106686592102,
698
+ "eval_mean_token_accuracy": 0.8465680112672407,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.2971,
701
+ "eval_samples_per_second": 15.933,
702
+ "eval_steps_per_second": 1.993,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5665927153080702,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.5987663865089417,
709
+ "learning_rate": 0.00022302249611363625,
710
+ "loss": 0.5143643856048584,
711
+ "mean_token_accuracy": 0.8529589556157589,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.568248552118623,
718
+ "eval_loss": 0.5476346015930176,
719
+ "eval_mean_token_accuracy": 0.8476775434128073,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.9583,
722
+ "eval_samples_per_second": 15.812,
723
+ "eval_steps_per_second": 1.978,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5673687808215618,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.735261857509613,
730
+ "learning_rate": 0.00022253790043595193,
731
+ "loss": 0.509885597229004,
732
+ "mean_token_accuracy": 0.8537046857178211,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5616967284748721,
739
+ "eval_loss": 0.5439274311065674,
740
+ "eval_mean_token_accuracy": 0.8488946217437123,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.0604,
743
+ "eval_samples_per_second": 15.977,
744
+ "eval_steps_per_second": 1.999,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.5529541682451964,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7014835476875305,
751
+ "learning_rate": 0.00022202031176171442,
752
+ "loss": 0.5078992366790771,
753
+ "mean_token_accuracy": 0.8525233261287213,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5827173320359962,
760
+ "eval_loss": 0.5419450402259827,
761
+ "eval_mean_token_accuracy": 0.8477318609176681,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 85.2984,
764
+ "eval_samples_per_second": 16.12,
765
+ "eval_steps_per_second": 2.016,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5755720350891351,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.705613911151886,
772
+ "learning_rate": 0.00022146988618860824,
773
+ "loss": 0.5181350708007812,
774
+ "mean_token_accuracy": 0.8467609457671642,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5743971356125765,
781
+ "eval_loss": 0.5415896773338318,
782
+ "eval_mean_token_accuracy": 0.847328585940738,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 85.5602,
785
+ "eval_samples_per_second": 16.071,
786
+ "eval_steps_per_second": 2.01,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.561330484598875,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.6722865700721741,
793
+ "learning_rate": 0.0002208867897174789,
794
+ "loss": 0.499837589263916,
795
+ "mean_token_accuracy": 0.8518734864890576,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.5865232653396074,
802
+ "eval_loss": 0.5437926650047302,
803
+ "eval_mean_token_accuracy": 0.8450997017843779,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4116,
806
+ "eval_samples_per_second": 15.912,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.547389242425561,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.7935577034950256,
814
+ "learning_rate": 0.00022027119820226907,
815
+ "loss": 0.4977591514587402,
816
+ "mean_token_accuracy": 0.8539491161704064,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.5290903090391048,
823
+ "eval_loss": 0.5409526824951172,
824
+ "eval_mean_token_accuracy": 0.8497545698354411,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.7262,
827
+ "eval_samples_per_second": 15.854,
828
+ "eval_steps_per_second": 1.983,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5687909748405218,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6180546283721924,
835
+ "learning_rate": 0.00021962329729698345,
836
+ "loss": 0.5109643459320068,
837
+ "mean_token_accuracy": 0.8521598495543004,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.5503541858390321,
844
+ "eval_loss": 0.5361555218696594,
845
+ "eval_mean_token_accuracy": 0.8510884285666221,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.3339,
848
+ "eval_samples_per_second": 15.927,
849
+ "eval_steps_per_second": 1.992,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4739728841261986,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.8058829307556152,
856
+ "learning_rate": 0.0002189432823996982,
857
+ "loss": 0.4204097747802734,
858
+ "mean_token_accuracy": 0.8728981889211215,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.5077334992414297,
865
+ "eval_loss": 0.5531114339828491,
866
+ "eval_mean_token_accuracy": 0.8489257208136625,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.4801,
869
+ "eval_samples_per_second": 15.9,
870
+ "eval_steps_per_second": 1.989,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.4594309840351343,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.6906896829605103,
877
+ "learning_rate": 0.0002182313585936314,
878
+ "loss": 0.4071959495544434,
879
+ "mean_token_accuracy": 0.8732857562601566,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.49850136994622474,
886
+ "eval_loss": 0.5486204624176025,
887
+ "eval_mean_token_accuracy": 0.8507991450470548,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.3364,
890
+ "eval_samples_per_second": 15.926,
891
+ "eval_steps_per_second": 1.992,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4881629109382629,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6343470215797424,
898
+ "learning_rate": 0.0002174877405852928,
899
+ "loss": 0.41669540405273436,
900
+ "mean_token_accuracy": 0.8711295068264008,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.49155513924914734,
907
+ "eval_loss": 0.555109441280365,
908
+ "eval_mean_token_accuracy": 0.8496399400539176,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 86.3295,
911
+ "eval_samples_per_second": 15.927,
912
+ "eval_steps_per_second": 1.992,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.4648668970912695,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.8014165163040161,
919
+ "learning_rate": 0.00021671265263973133,
920
+ "loss": 0.4110250473022461,
921
+ "mean_token_accuracy": 0.8754166305065155,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.4909258722219356,
928
+ "eval_loss": 0.5539511442184448,
929
+ "eval_mean_token_accuracy": 0.8492401502160138,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.3468,
932
+ "eval_samples_per_second": 15.924,
933
+ "eval_steps_per_second": 1.992,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4824485514312983,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6665191054344177,
940
+ "learning_rate": 0.00021590632851289967,
941
+ "loss": 0.4181404113769531,
942
+ "mean_token_accuracy": 0.8726993151009083,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.4986876940657926,
949
+ "eval_loss": 0.547695517539978,
950
+ "eval_mean_token_accuracy": 0.8501384708770486,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.3838,
953
+ "eval_samples_per_second": 15.917,
954
+ "eval_steps_per_second": 1.991,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.4751896943897009,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.81158047914505,
961
+ "learning_rate": 0.00021506901138115678,
962
+ "loss": 0.40689678192138673,
963
+ "mean_token_accuracy": 0.8745221219956875,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.507153491121392,
970
+ "eval_loss": 0.5501641631126404,
971
+ "eval_mean_token_accuracy": 0.8495670116918032,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.0912,
974
+ "eval_samples_per_second": 15.971,
975
+ "eval_steps_per_second": 1.998,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4873133715242147,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7218056321144104,
982
+ "learning_rate": 0.0002142009537679292,
983
+ "loss": 0.42701358795166017,
984
+ "mean_token_accuracy": 0.8695114746689796,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5202612736543943,
991
+ "eval_loss": 0.5491839051246643,
992
+ "eval_mean_token_accuracy": 0.8494071208460386,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.1142,
995
+ "eval_samples_per_second": 15.967,
996
+ "eval_steps_per_second": 1.997,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.4762951169162989,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 0.7194424867630005,
1003
+ "learning_rate": 0.0002133024174675534,
1004
+ "loss": 0.42299847602844237,
1005
+ "mean_token_accuracy": 0.8709790132939815,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4899340462546016,
1012
+ "eval_loss": 0.5522511601448059,
1013
+ "eval_mean_token_accuracy": 0.8492208258357159,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.463,
1016
+ "eval_samples_per_second": 15.903,
1017
+ "eval_steps_per_second": 1.989,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.49650347977876663,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.8406022787094116,
1024
+ "learning_rate": 0.0002123736734663221,
1025
+ "loss": 0.4275330066680908,
1026
+ "mean_token_accuracy": 0.8670595556497573,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.49691385654515996,
1033
+ "eval_loss": 0.5491269826889038,
1034
+ "eval_mean_token_accuracy": 0.850309816210769,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.17,
1037
+ "eval_samples_per_second": 15.957,
1038
+ "eval_steps_per_second": 1.996,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.48843890577554705,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.9082473516464233,
1045
+ "learning_rate": 0.00021141500186075868,
1046
+ "loss": 0.4309722423553467,
1047
+ "mean_token_accuracy": 0.8686766296625137,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5543508351195691,
1054
+ "eval_loss": 0.5478800535202026,
1055
+ "eval_mean_token_accuracy": 0.8478029522784921,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.3835,
1058
+ "eval_samples_per_second": 15.917,
1059
+ "eval_steps_per_second": 1.991,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.4777219031006098,
1064
+ "epoch": 2.5379825653798256,
1065
+ "grad_norm": 0.7448089122772217,
1066
+ "learning_rate": 0.0002104266917731438,
1067
+ "loss": 0.423325252532959,
1068
+ "mean_token_accuracy": 0.8706337086856365,
1069
+ "num_tokens": 2384270.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.5379825653798256,
1074
+ "eval_entropy": 0.49857561550168106,
1075
+ "eval_loss": 0.5511948466300964,
1076
+ "eval_mean_token_accuracy": 0.8502220289651737,
1077
+ "eval_num_tokens": 2384270.0,
1078
+ "eval_runtime": 86.5399,
1079
+ "eval_samples_per_second": 15.889,
1080
+ "eval_steps_per_second": 1.988,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.4844174191355705,
1085
+ "epoch": 2.587795765877958,
1086
+ "grad_norm": 0.794029176235199,
1087
+ "learning_rate": 0.00020940904126432,
1088
+ "loss": 0.4176753044128418,
1089
+ "mean_token_accuracy": 0.873535567522049,
1090
+ "num_tokens": 2428036.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.587795765877958,
1095
+ "eval_entropy": 0.485467542222766,
1096
+ "eval_loss": 0.5539286732673645,
1097
+ "eval_mean_token_accuracy": 0.8495475081510322,
1098
+ "eval_num_tokens": 2428036.0,
1099
+ "eval_runtime": 86.135,
1100
+ "eval_samples_per_second": 15.963,
1101
+ "eval_steps_per_second": 1.997,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.49070929251611234,
1106
+ "epoch": 2.6376089663760895,
1107
+ "grad_norm": 0.7558256983757019,
1108
+ "learning_rate": 0.0002083623572438007,
1109
+ "loss": 0.42867293357849123,
1110
+ "mean_token_accuracy": 0.8696666076779366,
1111
+ "num_tokens": 2476815.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.6376089663760895,
1116
+ "eval_entropy": 0.490822730889154,
1117
+ "eval_loss": 0.5434785485267639,
1118
+ "eval_mean_token_accuracy": 0.850568296950917,
1119
+ "eval_num_tokens": 2476815.0,
1120
+ "eval_runtime": 86.4933,
1121
+ "eval_samples_per_second": 15.897,
1122
+ "eval_steps_per_second": 1.989,
1123
+ "step": 1060
1124
+ },
1125
+ {
1126
+ "entropy": 0.47806114703416824,
1127
+ "epoch": 2.6874221668742218,
1128
+ "grad_norm": 0.6608979105949402,
1129
+ "learning_rate": 0.00020728695537721047,
1130
+ "loss": 0.4289727687835693,
1131
+ "mean_token_accuracy": 0.8693130135536193,
1132
+ "num_tokens": 2527131.0,
1133
+ "step": 1080
1134
+ },
1135
+ {
1136
+ "epoch": 2.6874221668742218,
1137
+ "eval_entropy": 0.5285773256490397,
1138
+ "eval_loss": 0.5444230437278748,
1139
+ "eval_mean_token_accuracy": 0.8498796481032704,
1140
+ "eval_num_tokens": 2527131.0,
1141
+ "eval_runtime": 86.7091,
1142
+ "eval_samples_per_second": 15.858,
1143
+ "eval_steps_per_second": 1.984,
1144
+ "step": 1080
1145
+ }
1146
+ ],
1147
+ "logging_steps": 20,
1148
+ "max_steps": 4020,
1149
+ "num_input_tokens_seen": 0,
1150
+ "num_train_epochs": 10,
1151
+ "save_steps": 20,
1152
+ "stateful_callbacks": {
1153
+ "TrainerControl": {
1154
+ "args": {
1155
+ "should_epoch_stop": false,
1156
+ "should_evaluate": false,
1157
+ "should_log": false,
1158
+ "should_save": true,
1159
+ "should_training_stop": false
1160
+ },
1161
+ "attributes": {}
1162
+ }
1163
+ },
1164
+ "total_flos": 1.0682640451304448e+17,
1165
+ "train_batch_size": 4,
1166
+ "trial_name": null,
1167
+ "trial_params": null
1168
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test2/checkpoint-1100/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.028265386974777595,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/README.md ADDED
@@ -0,0 +1,58 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: transformers
4
+ model_name: Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1
5
+ tags:
6
+ - generated_from_trainer
7
+ - sft
8
+ - trl
9
+ licence: license
10
+ ---
11
+
12
+ # Model Card for Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1
13
+
14
+ This model is a fine-tuned version of [Qwen/Qwen3.5-4B-Base](https://huggingface.co/Qwen/Qwen3.5-4B-Base).
15
+ It has been trained using [TRL](https://github.com/huggingface/trl).
16
+
17
+ ## Quick start
18
+
19
+ ```python
20
+ from transformers import pipeline
21
+
22
+ question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?"
23
+ generator = pipeline("text-generation", model="None", device="cuda")
24
+ output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0]
25
+ print(output["generated_text"])
26
+ ```
27
+
28
+ ## Training procedure
29
+
30
+ [<img src="https://raw.githubusercontent.com/wandb/assets/main/wandb-github-badge-28.svg" alt="Visualize in Weights & Biases" width="150" height="24"/>](https://wandb.ai/katriin-kukk/Cross_lingual_morphological_generalization/runs/5khkq4dz)
31
+
32
+
33
+
34
+ This model was trained with SFT.
35
+
36
+ ### Framework versions
37
+
38
+ - TRL: 0.29.0
39
+ - Transformers: 5.5.4
40
+ - Pytorch: 2.10.0
41
+ - Datasets: 4.6.1
42
+ - Tokenizers: 0.22.2
43
+
44
+ ## Citations
45
+
46
+
47
+
48
+ Cite TRL as:
49
+
50
+ ```bibtex
51
+ @software{vonwerra2020trl,
52
+ title = {{TRL: Transformers Reinforcement Learning}},
53
+ author = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin},
54
+ license = {Apache-2.0},
55
+ url = {https://github.com/huggingface/trl},
56
+ year = {2020}
57
+ }
58
+ ```
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1266/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1266/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.04758632698976937,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "k_proj",
33
+ "o_proj",
34
+ "gate_proj",
35
+ "v_proj",
36
+ "down_proj",
37
+ "up_proj",
38
+ "q_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1266/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1266/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1266/trainer_state.json ADDED
@@ -0,0 +1,317 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 3.0,
6
+ "eval_steps": 500,
7
+ "global_step": 1266,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 2.4865781700611116,
14
+ "epoch": 0.11869436201780416,
15
+ "grad_norm": 3.600203037261963,
16
+ "learning_rate": 9.55350170221182e-06,
17
+ "loss": 2.2448199462890623,
18
+ "mean_token_accuracy": 0.571574572622776,
19
+ "num_tokens": 66201.0,
20
+ "step": 50
21
+ },
22
+ {
23
+ "entropy": 1.1757256650924683,
24
+ "epoch": 0.23738872403560832,
25
+ "grad_norm": 1.843338966369629,
26
+ "learning_rate": 1.9301972826917757e-05,
27
+ "loss": 1.0319316864013672,
28
+ "mean_token_accuracy": 0.7366894924640656,
29
+ "num_tokens": 132943.0,
30
+ "step": 100
31
+ },
32
+ {
33
+ "entropy": 1.0132716038823129,
34
+ "epoch": 0.3560830860534125,
35
+ "grad_norm": 2.0066328048706055,
36
+ "learning_rate": 2.9050443951623695e-05,
37
+ "loss": 0.8726716613769532,
38
+ "mean_token_accuracy": 0.7669131025671959,
39
+ "num_tokens": 202797.0,
40
+ "step": 150
41
+ },
42
+ {
43
+ "entropy": 0.9535812222957611,
44
+ "epoch": 0.47477744807121663,
45
+ "grad_norm": 1.8880268335342407,
46
+ "learning_rate": 3.8798915076329635e-05,
47
+ "loss": 0.8159033966064453,
48
+ "mean_token_accuracy": 0.778084682226181,
49
+ "num_tokens": 266514.0,
50
+ "step": 200
51
+ },
52
+ {
53
+ "entropy": 0.9149203181266785,
54
+ "epoch": 0.5934718100890207,
55
+ "grad_norm": 1.5609782934188843,
56
+ "learning_rate": 4.8547386201035576e-05,
57
+ "loss": 0.7810882568359375,
58
+ "mean_token_accuracy": 0.7836975249648094,
59
+ "num_tokens": 332781.0,
60
+ "step": 250
61
+ },
62
+ {
63
+ "entropy": 0.9015455979108811,
64
+ "epoch": 0.712166172106825,
65
+ "grad_norm": 1.6056852340698242,
66
+ "learning_rate": 5.829585732574152e-05,
67
+ "loss": 0.7620333099365234,
68
+ "mean_token_accuracy": 0.7867170104384422,
69
+ "num_tokens": 396012.0,
70
+ "step": 300
71
+ },
72
+ {
73
+ "entropy": 0.8751798084378243,
74
+ "epoch": 0.8308605341246291,
75
+ "grad_norm": 1.457392930984497,
76
+ "learning_rate": 6.804432845044745e-05,
77
+ "loss": 0.7453135681152344,
78
+ "mean_token_accuracy": 0.7934546408057213,
79
+ "num_tokens": 464261.0,
80
+ "step": 350
81
+ },
82
+ {
83
+ "entropy": 0.8582844731211662,
84
+ "epoch": 0.9495548961424333,
85
+ "grad_norm": 1.325522541999817,
86
+ "learning_rate": 7.779279957515339e-05,
87
+ "loss": 0.7352320861816406,
88
+ "mean_token_accuracy": 0.7950925189256668,
89
+ "num_tokens": 531313.0,
90
+ "step": 400
91
+ },
92
+ {
93
+ "epoch": 1.0,
94
+ "eval_entropy": 0.699295549364815,
95
+ "eval_loss": 0.6675883531570435,
96
+ "eval_mean_token_accuracy": 0.8100697049620555,
97
+ "eval_num_tokens": 559577.0,
98
+ "eval_runtime": 113.6401,
99
+ "eval_samples_per_second": 12.003,
100
+ "eval_steps_per_second": 1.505,
101
+ "step": 422
102
+ },
103
+ {
104
+ "entropy": 0.8327096131852436,
105
+ "epoch": 1.0664688427299702,
106
+ "grad_norm": 0.9726872444152832,
107
+ "learning_rate": 8.226683697984623e-05,
108
+ "loss": 0.6989453887939453,
109
+ "mean_token_accuracy": 0.7999463737918641,
110
+ "num_tokens": 595986.0,
111
+ "step": 450
112
+ },
113
+ {
114
+ "entropy": 0.7919371470808982,
115
+ "epoch": 1.1851632047477745,
116
+ "grad_norm": 1.04917311668396,
117
+ "learning_rate": 8.219368143379697e-05,
118
+ "loss": 0.6691493225097657,
119
+ "mean_token_accuracy": 0.8088096314668656,
120
+ "num_tokens": 663734.0,
121
+ "step": 500
122
+ },
123
+ {
124
+ "entropy": 0.7981289568543434,
125
+ "epoch": 1.3038575667655787,
126
+ "grad_norm": 1.1144819259643555,
127
+ "learning_rate": 8.205030999972083e-05,
128
+ "loss": 0.6692163848876953,
129
+ "mean_token_accuracy": 0.8067614835500717,
130
+ "num_tokens": 730600.0,
131
+ "step": 550
132
+ },
133
+ {
134
+ "entropy": 0.7944329422712326,
135
+ "epoch": 1.4225519287833828,
136
+ "grad_norm": 1.017581820487976,
137
+ "learning_rate": 8.183696788331456e-05,
138
+ "loss": 0.6635546112060546,
139
+ "mean_token_accuracy": 0.8121234861016273,
140
+ "num_tokens": 793502.0,
141
+ "step": 600
142
+ },
143
+ {
144
+ "entropy": 0.7757732102274895,
145
+ "epoch": 1.5412462908011868,
146
+ "grad_norm": 0.9428858160972595,
147
+ "learning_rate": 8.155401995992886e-05,
148
+ "loss": 0.6541387939453125,
149
+ "mean_token_accuracy": 0.8148220491409301,
150
+ "num_tokens": 860956.0,
151
+ "step": 650
152
+ },
153
+ {
154
+ "entropy": 0.7693249759078026,
155
+ "epoch": 1.659940652818991,
156
+ "grad_norm": 0.8914806842803955,
157
+ "learning_rate": 8.120195015052839e-05,
158
+ "loss": 0.6372745132446289,
159
+ "mean_token_accuracy": 0.8154754737019538,
160
+ "num_tokens": 928211.0,
161
+ "step": 700
162
+ },
163
+ {
164
+ "entropy": 0.7677849313616752,
165
+ "epoch": 1.7786350148367953,
166
+ "grad_norm": 1.2037655115127563,
167
+ "learning_rate": 8.078136059405015e-05,
168
+ "loss": 0.6468383026123047,
169
+ "mean_token_accuracy": 0.8125368970632553,
170
+ "num_tokens": 993921.0,
171
+ "step": 750
172
+ },
173
+ {
174
+ "entropy": 0.7699458369612694,
175
+ "epoch": 1.8973293768545996,
176
+ "grad_norm": 0.8997814059257507,
177
+ "learning_rate": 8.02929706175755e-05,
178
+ "loss": 0.6416602325439453,
179
+ "mean_token_accuracy": 0.8159067538380623,
180
+ "num_tokens": 1060568.0,
181
+ "step": 800
182
+ },
183
+ {
184
+ "epoch": 2.0,
185
+ "eval_entropy": 0.6038547956455521,
186
+ "eval_loss": 0.6323259472846985,
187
+ "eval_mean_token_accuracy": 0.8163558896522076,
188
+ "eval_num_tokens": 1119154.0,
189
+ "eval_runtime": 111.6364,
190
+ "eval_samples_per_second": 12.218,
191
+ "eval_steps_per_second": 1.532,
192
+ "step": 844
193
+ },
194
+ {
195
+ "entropy": 0.7529444349598764,
196
+ "epoch": 2.0142433234421366,
197
+ "grad_norm": 0.9705535173416138,
198
+ "learning_rate": 7.973761550607747e-05,
199
+ "loss": 0.6287842178344727,
200
+ "mean_token_accuracy": 0.8173542984851121,
201
+ "num_tokens": 1127140.0,
202
+ "step": 850
203
+ },
204
+ {
205
+ "entropy": 0.6624361242353917,
206
+ "epoch": 2.1329376854599404,
207
+ "grad_norm": 1.0336796045303345,
208
+ "learning_rate": 7.911624507384729e-05,
209
+ "loss": 0.5305244064331055,
210
+ "mean_token_accuracy": 0.8395491230487824,
211
+ "num_tokens": 1192314.0,
212
+ "step": 900
213
+ },
214
+ {
215
+ "entropy": 0.6706090711057187,
216
+ "epoch": 2.2516320474777447,
217
+ "grad_norm": 1.1563575267791748,
218
+ "learning_rate": 7.842992204004328e-05,
219
+ "loss": 0.5347850036621093,
220
+ "mean_token_accuracy": 0.8390352365374565,
221
+ "num_tokens": 1257658.0,
222
+ "step": 950
223
+ },
224
+ {
225
+ "entropy": 0.6641162340342999,
226
+ "epoch": 2.370326409495549,
227
+ "grad_norm": 1.0999572277069092,
228
+ "learning_rate": 7.767982021114064e-05,
229
+ "loss": 0.5343616867065429,
230
+ "mean_token_accuracy": 0.8390876743197441,
231
+ "num_tokens": 1324523.0,
232
+ "step": 1000
233
+ },
234
+ {
235
+ "entropy": 0.6628478536009789,
236
+ "epoch": 2.489020771513353,
237
+ "grad_norm": 1.0276364088058472,
238
+ "learning_rate": 7.68672224733903e-05,
239
+ "loss": 0.5415428161621094,
240
+ "mean_token_accuracy": 0.8387553268671035,
241
+ "num_tokens": 1391419.0,
242
+ "step": 1050
243
+ },
244
+ {
245
+ "entropy": 0.671518052071333,
246
+ "epoch": 2.6077151335311575,
247
+ "grad_norm": 0.9451322555541992,
248
+ "learning_rate": 7.599351859872084e-05,
249
+ "loss": 0.5410358810424805,
250
+ "mean_token_accuracy": 0.8373630735278129,
251
+ "num_tokens": 1456470.0,
252
+ "step": 1100
253
+ },
254
+ {
255
+ "entropy": 0.6752241159975528,
256
+ "epoch": 2.7264094955489613,
257
+ "grad_norm": 0.8325166702270508,
258
+ "learning_rate": 7.506020286783527e-05,
259
+ "loss": 0.5409298706054687,
260
+ "mean_token_accuracy": 0.8369611689448356,
261
+ "num_tokens": 1523984.0,
262
+ "step": 1150
263
+ },
264
+ {
265
+ "entropy": 0.6462954029440879,
266
+ "epoch": 2.8451038575667655,
267
+ "grad_norm": 1.0445443391799927,
268
+ "learning_rate": 7.406887151456858e-05,
269
+ "loss": 0.5271347427368164,
270
+ "mean_token_accuracy": 0.8404733729362488,
271
+ "num_tokens": 1591962.0,
272
+ "step": 1200
273
+ },
274
+ {
275
+ "entropy": 0.6678441441059113,
276
+ "epoch": 2.96379821958457,
277
+ "grad_norm": 0.9832372665405273,
278
+ "learning_rate": 7.302121999587646e-05,
279
+ "loss": 0.537381706237793,
280
+ "mean_token_accuracy": 0.8383750656247139,
281
+ "num_tokens": 1658478.0,
282
+ "step": 1250
283
+ },
284
+ {
285
+ "epoch": 3.0,
286
+ "eval_entropy": 0.524203968675513,
287
+ "eval_loss": 0.6435813307762146,
288
+ "eval_mean_token_accuracy": 0.8187350073055915,
289
+ "eval_num_tokens": 1678731.0,
290
+ "eval_runtime": 111.7182,
291
+ "eval_samples_per_second": 12.209,
292
+ "eval_steps_per_second": 1.531,
293
+ "step": 1266
294
+ }
295
+ ],
296
+ "logging_steps": 50,
297
+ "max_steps": 4220,
298
+ "num_input_tokens_seen": 0,
299
+ "num_train_epochs": 10,
300
+ "save_steps": 500,
301
+ "stateful_callbacks": {
302
+ "TrainerControl": {
303
+ "args": {
304
+ "should_epoch_stop": false,
305
+ "should_evaluate": false,
306
+ "should_log": false,
307
+ "should_save": true,
308
+ "should_training_stop": false
309
+ },
310
+ "attributes": {}
311
+ }
312
+ },
313
+ "total_flos": 6.244685217283277e+16,
314
+ "train_batch_size": 4,
315
+ "trial_name": null,
316
+ "trial_params": null
317
+ }
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1688/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1688/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.04758632698976937,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "k_proj",
33
+ "o_proj",
34
+ "gate_proj",
35
+ "v_proj",
36
+ "down_proj",
37
+ "up_proj",
38
+ "q_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1688/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1688/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-1688/trainer_state.json ADDED
@@ -0,0 +1,408 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 4.0,
6
+ "eval_steps": 500,
7
+ "global_step": 1688,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 2.4865781700611116,
14
+ "epoch": 0.11869436201780416,
15
+ "grad_norm": 3.600203037261963,
16
+ "learning_rate": 9.55350170221182e-06,
17
+ "loss": 2.2448199462890623,
18
+ "mean_token_accuracy": 0.571574572622776,
19
+ "num_tokens": 66201.0,
20
+ "step": 50
21
+ },
22
+ {
23
+ "entropy": 1.1757256650924683,
24
+ "epoch": 0.23738872403560832,
25
+ "grad_norm": 1.843338966369629,
26
+ "learning_rate": 1.9301972826917757e-05,
27
+ "loss": 1.0319316864013672,
28
+ "mean_token_accuracy": 0.7366894924640656,
29
+ "num_tokens": 132943.0,
30
+ "step": 100
31
+ },
32
+ {
33
+ "entropy": 1.0132716038823129,
34
+ "epoch": 0.3560830860534125,
35
+ "grad_norm": 2.0066328048706055,
36
+ "learning_rate": 2.9050443951623695e-05,
37
+ "loss": 0.8726716613769532,
38
+ "mean_token_accuracy": 0.7669131025671959,
39
+ "num_tokens": 202797.0,
40
+ "step": 150
41
+ },
42
+ {
43
+ "entropy": 0.9535812222957611,
44
+ "epoch": 0.47477744807121663,
45
+ "grad_norm": 1.8880268335342407,
46
+ "learning_rate": 3.8798915076329635e-05,
47
+ "loss": 0.8159033966064453,
48
+ "mean_token_accuracy": 0.778084682226181,
49
+ "num_tokens": 266514.0,
50
+ "step": 200
51
+ },
52
+ {
53
+ "entropy": 0.9149203181266785,
54
+ "epoch": 0.5934718100890207,
55
+ "grad_norm": 1.5609782934188843,
56
+ "learning_rate": 4.8547386201035576e-05,
57
+ "loss": 0.7810882568359375,
58
+ "mean_token_accuracy": 0.7836975249648094,
59
+ "num_tokens": 332781.0,
60
+ "step": 250
61
+ },
62
+ {
63
+ "entropy": 0.9015455979108811,
64
+ "epoch": 0.712166172106825,
65
+ "grad_norm": 1.6056852340698242,
66
+ "learning_rate": 5.829585732574152e-05,
67
+ "loss": 0.7620333099365234,
68
+ "mean_token_accuracy": 0.7867170104384422,
69
+ "num_tokens": 396012.0,
70
+ "step": 300
71
+ },
72
+ {
73
+ "entropy": 0.8751798084378243,
74
+ "epoch": 0.8308605341246291,
75
+ "grad_norm": 1.457392930984497,
76
+ "learning_rate": 6.804432845044745e-05,
77
+ "loss": 0.7453135681152344,
78
+ "mean_token_accuracy": 0.7934546408057213,
79
+ "num_tokens": 464261.0,
80
+ "step": 350
81
+ },
82
+ {
83
+ "entropy": 0.8582844731211662,
84
+ "epoch": 0.9495548961424333,
85
+ "grad_norm": 1.325522541999817,
86
+ "learning_rate": 7.779279957515339e-05,
87
+ "loss": 0.7352320861816406,
88
+ "mean_token_accuracy": 0.7950925189256668,
89
+ "num_tokens": 531313.0,
90
+ "step": 400
91
+ },
92
+ {
93
+ "epoch": 1.0,
94
+ "eval_entropy": 0.699295549364815,
95
+ "eval_loss": 0.6675883531570435,
96
+ "eval_mean_token_accuracy": 0.8100697049620555,
97
+ "eval_num_tokens": 559577.0,
98
+ "eval_runtime": 113.6401,
99
+ "eval_samples_per_second": 12.003,
100
+ "eval_steps_per_second": 1.505,
101
+ "step": 422
102
+ },
103
+ {
104
+ "entropy": 0.8327096131852436,
105
+ "epoch": 1.0664688427299702,
106
+ "grad_norm": 0.9726872444152832,
107
+ "learning_rate": 8.226683697984623e-05,
108
+ "loss": 0.6989453887939453,
109
+ "mean_token_accuracy": 0.7999463737918641,
110
+ "num_tokens": 595986.0,
111
+ "step": 450
112
+ },
113
+ {
114
+ "entropy": 0.7919371470808982,
115
+ "epoch": 1.1851632047477745,
116
+ "grad_norm": 1.04917311668396,
117
+ "learning_rate": 8.219368143379697e-05,
118
+ "loss": 0.6691493225097657,
119
+ "mean_token_accuracy": 0.8088096314668656,
120
+ "num_tokens": 663734.0,
121
+ "step": 500
122
+ },
123
+ {
124
+ "entropy": 0.7981289568543434,
125
+ "epoch": 1.3038575667655787,
126
+ "grad_norm": 1.1144819259643555,
127
+ "learning_rate": 8.205030999972083e-05,
128
+ "loss": 0.6692163848876953,
129
+ "mean_token_accuracy": 0.8067614835500717,
130
+ "num_tokens": 730600.0,
131
+ "step": 550
132
+ },
133
+ {
134
+ "entropy": 0.7944329422712326,
135
+ "epoch": 1.4225519287833828,
136
+ "grad_norm": 1.017581820487976,
137
+ "learning_rate": 8.183696788331456e-05,
138
+ "loss": 0.6635546112060546,
139
+ "mean_token_accuracy": 0.8121234861016273,
140
+ "num_tokens": 793502.0,
141
+ "step": 600
142
+ },
143
+ {
144
+ "entropy": 0.7757732102274895,
145
+ "epoch": 1.5412462908011868,
146
+ "grad_norm": 0.9428858160972595,
147
+ "learning_rate": 8.155401995992886e-05,
148
+ "loss": 0.6541387939453125,
149
+ "mean_token_accuracy": 0.8148220491409301,
150
+ "num_tokens": 860956.0,
151
+ "step": 650
152
+ },
153
+ {
154
+ "entropy": 0.7693249759078026,
155
+ "epoch": 1.659940652818991,
156
+ "grad_norm": 0.8914806842803955,
157
+ "learning_rate": 8.120195015052839e-05,
158
+ "loss": 0.6372745132446289,
159
+ "mean_token_accuracy": 0.8154754737019538,
160
+ "num_tokens": 928211.0,
161
+ "step": 700
162
+ },
163
+ {
164
+ "entropy": 0.7677849313616752,
165
+ "epoch": 1.7786350148367953,
166
+ "grad_norm": 1.2037655115127563,
167
+ "learning_rate": 8.078136059405015e-05,
168
+ "loss": 0.6468383026123047,
169
+ "mean_token_accuracy": 0.8125368970632553,
170
+ "num_tokens": 993921.0,
171
+ "step": 750
172
+ },
173
+ {
174
+ "entropy": 0.7699458369612694,
175
+ "epoch": 1.8973293768545996,
176
+ "grad_norm": 0.8997814059257507,
177
+ "learning_rate": 8.02929706175755e-05,
178
+ "loss": 0.6416602325439453,
179
+ "mean_token_accuracy": 0.8159067538380623,
180
+ "num_tokens": 1060568.0,
181
+ "step": 800
182
+ },
183
+ {
184
+ "epoch": 2.0,
185
+ "eval_entropy": 0.6038547956455521,
186
+ "eval_loss": 0.6323259472846985,
187
+ "eval_mean_token_accuracy": 0.8163558896522076,
188
+ "eval_num_tokens": 1119154.0,
189
+ "eval_runtime": 111.6364,
190
+ "eval_samples_per_second": 12.218,
191
+ "eval_steps_per_second": 1.532,
192
+ "step": 844
193
+ },
194
+ {
195
+ "entropy": 0.7529444349598764,
196
+ "epoch": 2.0142433234421366,
197
+ "grad_norm": 0.9705535173416138,
198
+ "learning_rate": 7.973761550607747e-05,
199
+ "loss": 0.6287842178344727,
200
+ "mean_token_accuracy": 0.8173542984851121,
201
+ "num_tokens": 1127140.0,
202
+ "step": 850
203
+ },
204
+ {
205
+ "entropy": 0.6624361242353917,
206
+ "epoch": 2.1329376854599404,
207
+ "grad_norm": 1.0336796045303345,
208
+ "learning_rate": 7.911624507384729e-05,
209
+ "loss": 0.5305244064331055,
210
+ "mean_token_accuracy": 0.8395491230487824,
211
+ "num_tokens": 1192314.0,
212
+ "step": 900
213
+ },
214
+ {
215
+ "entropy": 0.6706090711057187,
216
+ "epoch": 2.2516320474777447,
217
+ "grad_norm": 1.1563575267791748,
218
+ "learning_rate": 7.842992204004328e-05,
219
+ "loss": 0.5347850036621093,
220
+ "mean_token_accuracy": 0.8390352365374565,
221
+ "num_tokens": 1257658.0,
222
+ "step": 950
223
+ },
224
+ {
225
+ "entropy": 0.6641162340342999,
226
+ "epoch": 2.370326409495549,
227
+ "grad_norm": 1.0999572277069092,
228
+ "learning_rate": 7.767982021114064e-05,
229
+ "loss": 0.5343616867065429,
230
+ "mean_token_accuracy": 0.8390876743197441,
231
+ "num_tokens": 1324523.0,
232
+ "step": 1000
233
+ },
234
+ {
235
+ "entropy": 0.6628478536009789,
236
+ "epoch": 2.489020771513353,
237
+ "grad_norm": 1.0276364088058472,
238
+ "learning_rate": 7.68672224733903e-05,
239
+ "loss": 0.5415428161621094,
240
+ "mean_token_accuracy": 0.8387553268671035,
241
+ "num_tokens": 1391419.0,
242
+ "step": 1050
243
+ },
244
+ {
245
+ "entropy": 0.671518052071333,
246
+ "epoch": 2.6077151335311575,
247
+ "grad_norm": 0.9451322555541992,
248
+ "learning_rate": 7.599351859872084e-05,
249
+ "loss": 0.5410358810424805,
250
+ "mean_token_accuracy": 0.8373630735278129,
251
+ "num_tokens": 1456470.0,
252
+ "step": 1100
253
+ },
254
+ {
255
+ "entropy": 0.6752241159975528,
256
+ "epoch": 2.7264094955489613,
257
+ "grad_norm": 0.8325166702270508,
258
+ "learning_rate": 7.506020286783527e-05,
259
+ "loss": 0.5409298706054687,
260
+ "mean_token_accuracy": 0.8369611689448356,
261
+ "num_tokens": 1523984.0,
262
+ "step": 1150
263
+ },
264
+ {
265
+ "entropy": 0.6462954029440879,
266
+ "epoch": 2.8451038575667655,
267
+ "grad_norm": 1.0445443391799927,
268
+ "learning_rate": 7.406887151456858e-05,
269
+ "loss": 0.5271347427368164,
270
+ "mean_token_accuracy": 0.8404733729362488,
271
+ "num_tokens": 1591962.0,
272
+ "step": 1200
273
+ },
274
+ {
275
+ "entropy": 0.6678441441059113,
276
+ "epoch": 2.96379821958457,
277
+ "grad_norm": 0.9832372665405273,
278
+ "learning_rate": 7.302121999587646e-05,
279
+ "loss": 0.537381706237793,
280
+ "mean_token_accuracy": 0.8383750656247139,
281
+ "num_tokens": 1658478.0,
282
+ "step": 1250
283
+ },
284
+ {
285
+ "epoch": 3.0,
286
+ "eval_entropy": 0.524203968675513,
287
+ "eval_loss": 0.6435813307762146,
288
+ "eval_mean_token_accuracy": 0.8187350073055915,
289
+ "eval_num_tokens": 1678731.0,
290
+ "eval_runtime": 111.7182,
291
+ "eval_samples_per_second": 12.209,
292
+ "eval_steps_per_second": 1.531,
293
+ "step": 1266
294
+ },
295
+ {
296
+ "entropy": 0.5911264679758682,
297
+ "epoch": 3.080712166172107,
298
+ "grad_norm": 1.2129673957824707,
299
+ "learning_rate": 7.19190400921244e-05,
300
+ "loss": 0.44908695220947265,
301
+ "mean_token_accuracy": 0.8601035639114186,
302
+ "num_tokens": 1725154.0,
303
+ "step": 1300
304
+ },
305
+ {
306
+ "entropy": 0.5566366592049599,
307
+ "epoch": 3.199406528189911,
308
+ "grad_norm": 0.9602940678596497,
309
+ "learning_rate": 7.076421684263661e-05,
310
+ "loss": 0.4135295867919922,
311
+ "mean_token_accuracy": 0.8689188846945762,
312
+ "num_tokens": 1792562.0,
313
+ "step": 1350
314
+ },
315
+ {
316
+ "entropy": 0.5605658321082592,
317
+ "epoch": 3.318100890207715,
318
+ "grad_norm": 1.0783617496490479,
319
+ "learning_rate": 6.955872532174566e-05,
320
+ "loss": 0.41924549102783204,
321
+ "mean_token_accuracy": 0.8669222807884216,
322
+ "num_tokens": 1858017.0,
323
+ "step": 1400
324
+ },
325
+ {
326
+ "entropy": 0.5531888791918754,
327
+ "epoch": 3.436795252225519,
328
+ "grad_norm": 1.285948395729065,
329
+ "learning_rate": 6.830462726085685e-05,
330
+ "loss": 0.41391544342041015,
331
+ "mean_token_accuracy": 0.8701067119836807,
332
+ "num_tokens": 1925739.0,
333
+ "step": 1450
334
+ },
335
+ {
336
+ "entropy": 0.5541697943210602,
337
+ "epoch": 3.5554896142433234,
338
+ "grad_norm": 1.3446345329284668,
339
+ "learning_rate": 6.700406752230453e-05,
340
+ "loss": 0.42396705627441406,
341
+ "mean_token_accuracy": 0.8686600789427757,
342
+ "num_tokens": 1992896.0,
343
+ "step": 1500
344
+ },
345
+ {
346
+ "entropy": 0.5668759573996067,
347
+ "epoch": 3.6741839762611277,
348
+ "grad_norm": 1.1999047994613647,
349
+ "learning_rate": 6.565927043103079e-05,
350
+ "loss": 0.42777458190917966,
351
+ "mean_token_accuracy": 0.8663509142398834,
352
+ "num_tokens": 2057970.0,
353
+ "step": 1550
354
+ },
355
+ {
356
+ "entropy": 0.5705659487843513,
357
+ "epoch": 3.792878338278932,
358
+ "grad_norm": 1.1414515972137451,
359
+ "learning_rate": 6.427253597036095e-05,
360
+ "loss": 0.42880672454833985,
361
+ "mean_token_accuracy": 0.8653362435102463,
362
+ "num_tokens": 2123534.0,
363
+ "step": 1600
364
+ },
365
+ {
366
+ "entropy": 0.558414245545864,
367
+ "epoch": 3.9115727002967358,
368
+ "grad_norm": 1.3412097692489624,
369
+ "learning_rate": 6.284623584838158e-05,
370
+ "loss": 0.4282422256469727,
371
+ "mean_token_accuracy": 0.866187039911747,
372
+ "num_tokens": 2189979.0,
373
+ "step": 1650
374
+ },
375
+ {
376
+ "epoch": 4.0,
377
+ "eval_entropy": 0.4868207575633512,
378
+ "eval_loss": 0.6648371815681458,
379
+ "eval_mean_token_accuracy": 0.8163467548046893,
380
+ "eval_num_tokens": 2238308.0,
381
+ "eval_runtime": 111.72,
382
+ "eval_samples_per_second": 12.209,
383
+ "eval_steps_per_second": 1.531,
384
+ "step": 1688
385
+ }
386
+ ],
387
+ "logging_steps": 50,
388
+ "max_steps": 4220,
389
+ "num_input_tokens_seen": 0,
390
+ "num_train_epochs": 10,
391
+ "save_steps": 500,
392
+ "stateful_callbacks": {
393
+ "TrainerControl": {
394
+ "args": {
395
+ "should_epoch_stop": false,
396
+ "should_evaluate": false,
397
+ "should_log": false,
398
+ "should_save": true,
399
+ "should_training_stop": false
400
+ },
401
+ "attributes": {}
402
+ }
403
+ },
404
+ "total_flos": 8.33006276254679e+16,
405
+ "train_batch_size": 4,
406
+ "trial_name": null,
407
+ "trial_params": null
408
+ }
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-2110/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-2110/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.04758632698976937,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "k_proj",
33
+ "o_proj",
34
+ "gate_proj",
35
+ "v_proj",
36
+ "down_proj",
37
+ "up_proj",
38
+ "q_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-2110/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-2110/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
productivity_original_Estonian/Qwen3.5-4B-Base_productivity_splits_original_features_train_productivity_splits_original_features_test1/checkpoint-2110/trainer_state.json ADDED
@@ -0,0 +1,509 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 5.0,
6
+ "eval_steps": 500,
7
+ "global_step": 2110,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 2.4865781700611116,
14
+ "epoch": 0.11869436201780416,
15
+ "grad_norm": 3.600203037261963,
16
+ "learning_rate": 9.55350170221182e-06,
17
+ "loss": 2.2448199462890623,
18
+ "mean_token_accuracy": 0.571574572622776,
19
+ "num_tokens": 66201.0,
20
+ "step": 50
21
+ },
22
+ {
23
+ "entropy": 1.1757256650924683,
24
+ "epoch": 0.23738872403560832,
25
+ "grad_norm": 1.843338966369629,
26
+ "learning_rate": 1.9301972826917757e-05,
27
+ "loss": 1.0319316864013672,
28
+ "mean_token_accuracy": 0.7366894924640656,
29
+ "num_tokens": 132943.0,
30
+ "step": 100
31
+ },
32
+ {
33
+ "entropy": 1.0132716038823129,
34
+ "epoch": 0.3560830860534125,
35
+ "grad_norm": 2.0066328048706055,
36
+ "learning_rate": 2.9050443951623695e-05,
37
+ "loss": 0.8726716613769532,
38
+ "mean_token_accuracy": 0.7669131025671959,
39
+ "num_tokens": 202797.0,
40
+ "step": 150
41
+ },
42
+ {
43
+ "entropy": 0.9535812222957611,
44
+ "epoch": 0.47477744807121663,
45
+ "grad_norm": 1.8880268335342407,
46
+ "learning_rate": 3.8798915076329635e-05,
47
+ "loss": 0.8159033966064453,
48
+ "mean_token_accuracy": 0.778084682226181,
49
+ "num_tokens": 266514.0,
50
+ "step": 200
51
+ },
52
+ {
53
+ "entropy": 0.9149203181266785,
54
+ "epoch": 0.5934718100890207,
55
+ "grad_norm": 1.5609782934188843,
56
+ "learning_rate": 4.8547386201035576e-05,
57
+ "loss": 0.7810882568359375,
58
+ "mean_token_accuracy": 0.7836975249648094,
59
+ "num_tokens": 332781.0,
60
+ "step": 250
61
+ },
62
+ {
63
+ "entropy": 0.9015455979108811,
64
+ "epoch": 0.712166172106825,
65
+ "grad_norm": 1.6056852340698242,
66
+ "learning_rate": 5.829585732574152e-05,
67
+ "loss": 0.7620333099365234,
68
+ "mean_token_accuracy": 0.7867170104384422,
69
+ "num_tokens": 396012.0,
70
+ "step": 300
71
+ },
72
+ {
73
+ "entropy": 0.8751798084378243,
74
+ "epoch": 0.8308605341246291,
75
+ "grad_norm": 1.457392930984497,
76
+ "learning_rate": 6.804432845044745e-05,
77
+ "loss": 0.7453135681152344,
78
+ "mean_token_accuracy": 0.7934546408057213,
79
+ "num_tokens": 464261.0,
80
+ "step": 350
81
+ },
82
+ {
83
+ "entropy": 0.8582844731211662,
84
+ "epoch": 0.9495548961424333,
85
+ "grad_norm": 1.325522541999817,
86
+ "learning_rate": 7.779279957515339e-05,
87
+ "loss": 0.7352320861816406,
88
+ "mean_token_accuracy": 0.7950925189256668,
89
+ "num_tokens": 531313.0,
90
+ "step": 400
91
+ },
92
+ {
93
+ "epoch": 1.0,
94
+ "eval_entropy": 0.699295549364815,
95
+ "eval_loss": 0.6675883531570435,
96
+ "eval_mean_token_accuracy": 0.8100697049620555,
97
+ "eval_num_tokens": 559577.0,
98
+ "eval_runtime": 113.6401,
99
+ "eval_samples_per_second": 12.003,
100
+ "eval_steps_per_second": 1.505,
101
+ "step": 422
102
+ },
103
+ {
104
+ "entropy": 0.8327096131852436,
105
+ "epoch": 1.0664688427299702,
106
+ "grad_norm": 0.9726872444152832,
107
+ "learning_rate": 8.226683697984623e-05,
108
+ "loss": 0.6989453887939453,
109
+ "mean_token_accuracy": 0.7999463737918641,
110
+ "num_tokens": 595986.0,
111
+ "step": 450
112
+ },
113
+ {
114
+ "entropy": 0.7919371470808982,
115
+ "epoch": 1.1851632047477745,
116
+ "grad_norm": 1.04917311668396,
117
+ "learning_rate": 8.219368143379697e-05,
118
+ "loss": 0.6691493225097657,
119
+ "mean_token_accuracy": 0.8088096314668656,
120
+ "num_tokens": 663734.0,
121
+ "step": 500
122
+ },
123
+ {
124
+ "entropy": 0.7981289568543434,
125
+ "epoch": 1.3038575667655787,
126
+ "grad_norm": 1.1144819259643555,
127
+ "learning_rate": 8.205030999972083e-05,
128
+ "loss": 0.6692163848876953,
129
+ "mean_token_accuracy": 0.8067614835500717,
130
+ "num_tokens": 730600.0,
131
+ "step": 550
132
+ },
133
+ {
134
+ "entropy": 0.7944329422712326,
135
+ "epoch": 1.4225519287833828,
136
+ "grad_norm": 1.017581820487976,
137
+ "learning_rate": 8.183696788331456e-05,
138
+ "loss": 0.6635546112060546,
139
+ "mean_token_accuracy": 0.8121234861016273,
140
+ "num_tokens": 793502.0,
141
+ "step": 600
142
+ },
143
+ {
144
+ "entropy": 0.7757732102274895,
145
+ "epoch": 1.5412462908011868,
146
+ "grad_norm": 0.9428858160972595,
147
+ "learning_rate": 8.155401995992886e-05,
148
+ "loss": 0.6541387939453125,
149
+ "mean_token_accuracy": 0.8148220491409301,
150
+ "num_tokens": 860956.0,
151
+ "step": 650
152
+ },
153
+ {
154
+ "entropy": 0.7693249759078026,
155
+ "epoch": 1.659940652818991,
156
+ "grad_norm": 0.8914806842803955,
157
+ "learning_rate": 8.120195015052839e-05,
158
+ "loss": 0.6372745132446289,
159
+ "mean_token_accuracy": 0.8154754737019538,
160
+ "num_tokens": 928211.0,
161
+ "step": 700
162
+ },
163
+ {
164
+ "entropy": 0.7677849313616752,
165
+ "epoch": 1.7786350148367953,
166
+ "grad_norm": 1.2037655115127563,
167
+ "learning_rate": 8.078136059405015e-05,
168
+ "loss": 0.6468383026123047,
169
+ "mean_token_accuracy": 0.8125368970632553,
170
+ "num_tokens": 993921.0,
171
+ "step": 750
172
+ },
173
+ {
174
+ "entropy": 0.7699458369612694,
175
+ "epoch": 1.8973293768545996,
176
+ "grad_norm": 0.8997814059257507,
177
+ "learning_rate": 8.02929706175755e-05,
178
+ "loss": 0.6416602325439453,
179
+ "mean_token_accuracy": 0.8159067538380623,
180
+ "num_tokens": 1060568.0,
181
+ "step": 800
182
+ },
183
+ {
184
+ "epoch": 2.0,
185
+ "eval_entropy": 0.6038547956455521,
186
+ "eval_loss": 0.6323259472846985,
187
+ "eval_mean_token_accuracy": 0.8163558896522076,
188
+ "eval_num_tokens": 1119154.0,
189
+ "eval_runtime": 111.6364,
190
+ "eval_samples_per_second": 12.218,
191
+ "eval_steps_per_second": 1.532,
192
+ "step": 844
193
+ },
194
+ {
195
+ "entropy": 0.7529444349598764,
196
+ "epoch": 2.0142433234421366,
197
+ "grad_norm": 0.9705535173416138,
198
+ "learning_rate": 7.973761550607747e-05,
199
+ "loss": 0.6287842178344727,
200
+ "mean_token_accuracy": 0.8173542984851121,
201
+ "num_tokens": 1127140.0,
202
+ "step": 850
203
+ },
204
+ {
205
+ "entropy": 0.6624361242353917,
206
+ "epoch": 2.1329376854599404,
207
+ "grad_norm": 1.0336796045303345,
208
+ "learning_rate": 7.911624507384729e-05,
209
+ "loss": 0.5305244064331055,
210
+ "mean_token_accuracy": 0.8395491230487824,
211
+ "num_tokens": 1192314.0,
212
+ "step": 900
213
+ },
214
+ {
215
+ "entropy": 0.6706090711057187,
216
+ "epoch": 2.2516320474777447,
217
+ "grad_norm": 1.1563575267791748,
218
+ "learning_rate": 7.842992204004328e-05,
219
+ "loss": 0.5347850036621093,
220
+ "mean_token_accuracy": 0.8390352365374565,
221
+ "num_tokens": 1257658.0,
222
+ "step": 950
223
+ },
224
+ {
225
+ "entropy": 0.6641162340342999,
226
+ "epoch": 2.370326409495549,
227
+ "grad_norm": 1.0999572277069092,
228
+ "learning_rate": 7.767982021114064e-05,
229
+ "loss": 0.5343616867065429,
230
+ "mean_token_accuracy": 0.8390876743197441,
231
+ "num_tokens": 1324523.0,
232
+ "step": 1000
233
+ },
234
+ {
235
+ "entropy": 0.6628478536009789,
236
+ "epoch": 2.489020771513353,
237
+ "grad_norm": 1.0276364088058472,
238
+ "learning_rate": 7.68672224733903e-05,
239
+ "loss": 0.5415428161621094,
240
+ "mean_token_accuracy": 0.8387553268671035,
241
+ "num_tokens": 1391419.0,
242
+ "step": 1050
243
+ },
244
+ {
245
+ "entropy": 0.671518052071333,
246
+ "epoch": 2.6077151335311575,
247
+ "grad_norm": 0.9451322555541992,
248
+ "learning_rate": 7.599351859872084e-05,
249
+ "loss": 0.5410358810424805,
250
+ "mean_token_accuracy": 0.8373630735278129,
251
+ "num_tokens": 1456470.0,
252
+ "step": 1100
253
+ },
254
+ {
255
+ "entropy": 0.6752241159975528,
256
+ "epoch": 2.7264094955489613,
257
+ "grad_norm": 0.8325166702270508,
258
+ "learning_rate": 7.506020286783527e-05,
259
+ "loss": 0.5409298706054687,
260
+ "mean_token_accuracy": 0.8369611689448356,
261
+ "num_tokens": 1523984.0,
262
+ "step": 1150
263
+ },
264
+ {
265
+ "entropy": 0.6462954029440879,
266
+ "epoch": 2.8451038575667655,
267
+ "grad_norm": 1.0445443391799927,
268
+ "learning_rate": 7.406887151456858e-05,
269
+ "loss": 0.5271347427368164,
270
+ "mean_token_accuracy": 0.8404733729362488,
271
+ "num_tokens": 1591962.0,
272
+ "step": 1200
273
+ },
274
+ {
275
+ "entropy": 0.6678441441059113,
276
+ "epoch": 2.96379821958457,
277
+ "grad_norm": 0.9832372665405273,
278
+ "learning_rate": 7.302121999587646e-05,
279
+ "loss": 0.537381706237793,
280
+ "mean_token_accuracy": 0.8383750656247139,
281
+ "num_tokens": 1658478.0,
282
+ "step": 1250
283
+ },
284
+ {
285
+ "epoch": 3.0,
286
+ "eval_entropy": 0.524203968675513,
287
+ "eval_loss": 0.6435813307762146,
288
+ "eval_mean_token_accuracy": 0.8187350073055915,
289
+ "eval_num_tokens": 1678731.0,
290
+ "eval_runtime": 111.7182,
291
+ "eval_samples_per_second": 12.209,
292
+ "eval_steps_per_second": 1.531,
293
+ "step": 1266
294
+ },
295
+ {
296
+ "entropy": 0.5911264679758682,
297
+ "epoch": 3.080712166172107,
298
+ "grad_norm": 1.2129673957824707,
299
+ "learning_rate": 7.19190400921244e-05,
300
+ "loss": 0.44908695220947265,
301
+ "mean_token_accuracy": 0.8601035639114186,
302
+ "num_tokens": 1725154.0,
303
+ "step": 1300
304
+ },
305
+ {
306
+ "entropy": 0.5566366592049599,
307
+ "epoch": 3.199406528189911,
308
+ "grad_norm": 0.9602940678596497,
309
+ "learning_rate": 7.076421684263661e-05,
310
+ "loss": 0.4135295867919922,
311
+ "mean_token_accuracy": 0.8689188846945762,
312
+ "num_tokens": 1792562.0,
313
+ "step": 1350
314
+ },
315
+ {
316
+ "entropy": 0.5605658321082592,
317
+ "epoch": 3.318100890207715,
318
+ "grad_norm": 1.0783617496490479,
319
+ "learning_rate": 6.955872532174566e-05,
320
+ "loss": 0.41924549102783204,
321
+ "mean_token_accuracy": 0.8669222807884216,
322
+ "num_tokens": 1858017.0,
323
+ "step": 1400
324
+ },
325
+ {
326
+ "entropy": 0.5531888791918754,
327
+ "epoch": 3.436795252225519,
328
+ "grad_norm": 1.285948395729065,
329
+ "learning_rate": 6.830462726085685e-05,
330
+ "loss": 0.41391544342041015,
331
+ "mean_token_accuracy": 0.8701067119836807,
332
+ "num_tokens": 1925739.0,
333
+ "step": 1450
334
+ },
335
+ {
336
+ "entropy": 0.5541697943210602,
337
+ "epoch": 3.5554896142433234,
338
+ "grad_norm": 1.3446345329284668,
339
+ "learning_rate": 6.700406752230453e-05,
340
+ "loss": 0.42396705627441406,
341
+ "mean_token_accuracy": 0.8686600789427757,
342
+ "num_tokens": 1992896.0,
343
+ "step": 1500
344
+ },
345
+ {
346
+ "entropy": 0.5668759573996067,
347
+ "epoch": 3.6741839762611277,
348
+ "grad_norm": 1.1999047994613647,
349
+ "learning_rate": 6.565927043103079e-05,
350
+ "loss": 0.42777458190917966,
351
+ "mean_token_accuracy": 0.8663509142398834,
352
+ "num_tokens": 2057970.0,
353
+ "step": 1550
354
+ },
355
+ {
356
+ "entropy": 0.5705659487843513,
357
+ "epoch": 3.792878338278932,
358
+ "grad_norm": 1.1414515972137451,
359
+ "learning_rate": 6.427253597036095e-05,
360
+ "loss": 0.42880672454833985,
361
+ "mean_token_accuracy": 0.8653362435102463,
362
+ "num_tokens": 2123534.0,
363
+ "step": 1600
364
+ },
365
+ {
366
+ "entropy": 0.558414245545864,
367
+ "epoch": 3.9115727002967358,
368
+ "grad_norm": 1.3412097692489624,
369
+ "learning_rate": 6.284623584838158e-05,
370
+ "loss": 0.4282422256469727,
371
+ "mean_token_accuracy": 0.866187039911747,
372
+ "num_tokens": 2189979.0,
373
+ "step": 1650
374
+ },
375
+ {
376
+ "epoch": 4.0,
377
+ "eval_entropy": 0.4868207575633512,
378
+ "eval_loss": 0.6648371815681458,
379
+ "eval_mean_token_accuracy": 0.8163467548046893,
380
+ "eval_num_tokens": 2238308.0,
381
+ "eval_runtime": 111.72,
382
+ "eval_samples_per_second": 12.209,
383
+ "eval_steps_per_second": 1.531,
384
+ "step": 1688
385
+ },
386
+ {
387
+ "entropy": 0.546174580978258,
388
+ "epoch": 4.028486646884273,
389
+ "grad_norm": 1.464382290840149,
390
+ "learning_rate": 6.138280944164903e-05,
391
+ "loss": 0.40503074645996096,
392
+ "mean_token_accuracy": 0.8718915990161412,
393
+ "num_tokens": 2253247.0,
394
+ "step": 1700
395
+ },
396
+ {
397
+ "entropy": 0.45237040892243385,
398
+ "epoch": 4.147181008902077,
399
+ "grad_norm": 1.7151323556900024,
400
+ "learning_rate": 5.988475962316552e-05,
401
+ "loss": 0.3065692901611328,
402
+ "mean_token_accuracy": 0.900569304227829,
403
+ "num_tokens": 2320148.0,
404
+ "step": 1750
405
+ },
406
+ {
407
+ "entropy": 0.44834635987877847,
408
+ "epoch": 4.265875370919881,
409
+ "grad_norm": 1.2050637006759644,
410
+ "learning_rate": 5.835464848175874e-05,
411
+ "loss": 0.30684595108032225,
412
+ "mean_token_accuracy": 0.9003708437085152,
413
+ "num_tokens": 2386764.0,
414
+ "step": 1800
415
+ },
416
+ {
417
+ "entropy": 0.4549902780354023,
418
+ "epoch": 4.384569732937686,
419
+ "grad_norm": 1.20978844165802,
420
+ "learning_rate": 5.679509294018524e-05,
421
+ "loss": 0.3107210350036621,
422
+ "mean_token_accuracy": 0.8997164958715439,
423
+ "num_tokens": 2453770.0,
424
+ "step": 1850
425
+ },
426
+ {
427
+ "entropy": 0.46374980479478833,
428
+ "epoch": 4.503264094955489,
429
+ "grad_norm": 1.0553879737854004,
430
+ "learning_rate": 5.520876027945252e-05,
431
+ "loss": 0.3163416862487793,
432
+ "mean_token_accuracy": 0.8980184662342071,
433
+ "num_tokens": 2520198.0,
434
+ "step": 1900
435
+ },
436
+ {
437
+ "entropy": 0.4559279951453209,
438
+ "epoch": 4.621958456973294,
439
+ "grad_norm": 1.2723990678787231,
440
+ "learning_rate": 5.359836357701423e-05,
441
+ "loss": 0.31503250122070314,
442
+ "mean_token_accuracy": 0.8980488586425781,
443
+ "num_tokens": 2587420.0,
444
+ "step": 1950
445
+ },
446
+ {
447
+ "entropy": 0.45531487330794335,
448
+ "epoch": 4.740652818991098,
449
+ "grad_norm": 1.3452478647232056,
450
+ "learning_rate": 5.1966657066640514e-05,
451
+ "loss": 0.3135023880004883,
452
+ "mean_token_accuracy": 0.8982085168361664,
453
+ "num_tokens": 2652728.0,
454
+ "step": 2000
455
+ },
456
+ {
457
+ "entropy": 0.4609416849911213,
458
+ "epoch": 4.859347181008902,
459
+ "grad_norm": 1.37790846824646,
460
+ "learning_rate": 5.0316431427899296e-05,
461
+ "loss": 0.3144682502746582,
462
+ "mean_token_accuracy": 0.8983592641353607,
463
+ "num_tokens": 2718527.0,
464
+ "step": 2050
465
+ },
466
+ {
467
+ "entropy": 0.45369990602135657,
468
+ "epoch": 4.978041543026706,
469
+ "grad_norm": 1.4144543409347534,
470
+ "learning_rate": 4.865050901330515e-05,
471
+ "loss": 0.31526716232299806,
472
+ "mean_token_accuracy": 0.8976324373483657,
473
+ "num_tokens": 2785760.0,
474
+ "step": 2100
475
+ },
476
+ {
477
+ "epoch": 5.0,
478
+ "eval_entropy": 0.4259050552956542,
479
+ "eval_loss": 0.7592839002609253,
480
+ "eval_mean_token_accuracy": 0.8117133866973788,
481
+ "eval_num_tokens": 2797885.0,
482
+ "eval_runtime": 111.9255,
483
+ "eval_samples_per_second": 12.187,
484
+ "eval_steps_per_second": 1.528,
485
+ "step": 2110
486
+ }
487
+ ],
488
+ "logging_steps": 50,
489
+ "max_steps": 4220,
490
+ "num_input_tokens_seen": 0,
491
+ "num_train_epochs": 10,
492
+ "save_steps": 500,
493
+ "stateful_callbacks": {
494
+ "TrainerControl": {
495
+ "args": {
496
+ "should_epoch_stop": false,
497
+ "should_evaluate": false,
498
+ "should_log": false,
499
+ "should_save": true,
500
+ "should_training_stop": false
501
+ },
502
+ "attributes": {}
503
+ }
504
+ },
505
+ "total_flos": 1.0419291201852211e+17,
506
+ "train_batch_size": 4,
507
+ "trial_name": null,
508
+ "trial_params": null
509
+ }