K2triinK commited on
Commit
ce6a904
·
verified ·
1 Parent(s): 168922a

Add files using upload-large-folder tool

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-100/README.md +209 -0
  2. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-100/adapter_config.json +46 -0
  3. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-100/chat_template.jinja +154 -0
  4. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-100/tokenizer_config.json +31 -0
  5. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-100/trainer_state.json +139 -0
  6. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1000/README.md +209 -0
  7. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1000/adapter_config.json +46 -0
  8. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1000/chat_template.jinja +154 -0
  9. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1000/tokenizer_config.json +31 -0
  10. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1000/trainer_state.json +1084 -0
  11. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1020/README.md +209 -0
  12. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1020/adapter_config.json +46 -0
  13. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1020/chat_template.jinja +154 -0
  14. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1020/tokenizer_config.json +31 -0
  15. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1020/trainer_state.json +1105 -0
  16. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1040/README.md +209 -0
  17. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1040/adapter_config.json +46 -0
  18. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1040/chat_template.jinja +154 -0
  19. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1040/tokenizer_config.json +31 -0
  20. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1040/trainer_state.json +1126 -0
  21. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1060/README.md +209 -0
  22. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1060/adapter_config.json +46 -0
  23. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1060/chat_template.jinja +154 -0
  24. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1060/tokenizer_config.json +31 -0
  25. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1060/trainer_state.json +1147 -0
  26. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1080/README.md +209 -0
  27. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1080/adapter_config.json +46 -0
  28. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1080/chat_template.jinja +154 -0
  29. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1080/tokenizer_config.json +31 -0
  30. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1080/trainer_state.json +1168 -0
  31. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1100/README.md +209 -0
  32. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1100/adapter_config.json +46 -0
  33. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1100/chat_template.jinja +154 -0
  34. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1100/tokenizer_config.json +31 -0
  35. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1100/trainer_state.json +1189 -0
  36. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1120/README.md +209 -0
  37. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1120/adapter_config.json +46 -0
  38. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1120/chat_template.jinja +154 -0
  39. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1120/tokenizer_config.json +31 -0
  40. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1120/trainer_state.json +1210 -0
  41. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1140/README.md +209 -0
  42. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1140/adapter_config.json +46 -0
  43. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1140/chat_template.jinja +154 -0
  44. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1140/tokenizer_config.json +31 -0
  45. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1140/trainer_state.json +1231 -0
  46. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1160/README.md +209 -0
  47. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1160/adapter_config.json +46 -0
  48. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1160/chat_template.jinja +154 -0
  49. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1160/tokenizer_config.json +31 -0
  50. overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1160/trainer_state.json +1252 -0
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-100/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-100/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0005183818805460705,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-100/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-100/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-100/trainer_state.json ADDED
@@ -0,0 +1,139 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.24906600249066002,
6
+ "eval_steps": 20,
7
+ "global_step": 100,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.9784346982836722,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.0229668617248535,
16
+ "learning_rate": 9.526142962415369e-06,
17
+ "loss": 1.7360023498535155,
18
+ "mean_token_accuracy": 0.6449888605624438,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.41506897571475,
25
+ "eval_loss": 1.1876318454742432,
26
+ "eval_mean_token_accuracy": 0.734131895525511,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.8071,
29
+ "eval_samples_per_second": 15.671,
30
+ "eval_steps_per_second": 1.959,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.049924298375845,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.5795097351074219,
37
+ "learning_rate": 1.9553661870221022e-05,
38
+ "loss": 0.8944448471069336,
39
+ "mean_token_accuracy": 0.7748479396104813,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7996658658565476,
46
+ "eval_loss": 0.7202735543251038,
47
+ "eval_mean_token_accuracy": 0.8070558306089667,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.9199,
50
+ "eval_samples_per_second": 15.831,
51
+ "eval_steps_per_second": 1.979,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7734908878803253,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3136248588562012,
58
+ "learning_rate": 2.9581180778026673e-05,
59
+ "loss": 0.6780608654022217,
60
+ "mean_token_accuracy": 0.8168170280754566,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7119324009778888,
67
+ "eval_loss": 0.6554311513900757,
68
+ "eval_mean_token_accuracy": 0.8215604798738346,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.8692,
71
+ "eval_samples_per_second": 15.84,
72
+ "eval_steps_per_second": 1.98,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7071127541363239,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.387060284614563,
79
+ "learning_rate": 3.960869968583232e-05,
80
+ "loss": 0.6382100582122803,
81
+ "mean_token_accuracy": 0.8229366384446621,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6883931482254073,
88
+ "eval_loss": 0.625065803527832,
89
+ "eval_mean_token_accuracy": 0.828940509710201,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.662,
92
+ "eval_samples_per_second": 15.878,
93
+ "eval_steps_per_second": 1.985,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6800824083387852,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9892916679382324,
100
+ "learning_rate": 4.963621859363797e-05,
101
+ "loss": 0.6011715888977051,
102
+ "mean_token_accuracy": 0.8323964163661003,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6840810470802839,
109
+ "eval_loss": 0.6037028431892395,
110
+ "eval_mean_token_accuracy": 0.8309669033732525,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.4637,
113
+ "eval_samples_per_second": 15.914,
114
+ "eval_steps_per_second": 1.989,
115
+ "step": 100
116
+ }
117
+ ],
118
+ "logging_steps": 20,
119
+ "max_steps": 4020,
120
+ "num_input_tokens_seen": 0,
121
+ "num_train_epochs": 10,
122
+ "save_steps": 20,
123
+ "stateful_callbacks": {
124
+ "TrainerControl": {
125
+ "args": {
126
+ "should_epoch_stop": false,
127
+ "should_evaluate": false,
128
+ "should_log": false,
129
+ "should_save": true,
130
+ "should_training_stop": false
131
+ },
132
+ "attributes": {}
133
+ }
134
+ },
135
+ "total_flos": 9823576763965440.0,
136
+ "train_batch_size": 4,
137
+ "trial_name": null,
138
+ "trial_params": null
139
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1000/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1000/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0005183818805460705,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1000/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1000/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1000/trainer_state.json ADDED
@@ -0,0 +1,1084 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.488169364881694,
6
+ "eval_steps": 20,
7
+ "global_step": 1000,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.9784346982836722,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.0229668617248535,
16
+ "learning_rate": 9.526142962415369e-06,
17
+ "loss": 1.7360023498535155,
18
+ "mean_token_accuracy": 0.6449888605624438,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.41506897571475,
25
+ "eval_loss": 1.1876318454742432,
26
+ "eval_mean_token_accuracy": 0.734131895525511,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.8071,
29
+ "eval_samples_per_second": 15.671,
30
+ "eval_steps_per_second": 1.959,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.049924298375845,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.5795097351074219,
37
+ "learning_rate": 1.9553661870221022e-05,
38
+ "loss": 0.8944448471069336,
39
+ "mean_token_accuracy": 0.7748479396104813,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7996658658565476,
46
+ "eval_loss": 0.7202735543251038,
47
+ "eval_mean_token_accuracy": 0.8070558306089667,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.9199,
50
+ "eval_samples_per_second": 15.831,
51
+ "eval_steps_per_second": 1.979,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7734908878803253,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3136248588562012,
58
+ "learning_rate": 2.9581180778026673e-05,
59
+ "loss": 0.6780608654022217,
60
+ "mean_token_accuracy": 0.8168170280754566,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7119324009778888,
67
+ "eval_loss": 0.6554311513900757,
68
+ "eval_mean_token_accuracy": 0.8215604798738346,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.8692,
71
+ "eval_samples_per_second": 15.84,
72
+ "eval_steps_per_second": 1.98,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7071127541363239,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.387060284614563,
79
+ "learning_rate": 3.960869968583232e-05,
80
+ "loss": 0.6382100582122803,
81
+ "mean_token_accuracy": 0.8229366384446621,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6883931482254073,
88
+ "eval_loss": 0.625065803527832,
89
+ "eval_mean_token_accuracy": 0.828940509710201,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.662,
92
+ "eval_samples_per_second": 15.878,
93
+ "eval_steps_per_second": 1.985,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6800824083387852,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9892916679382324,
100
+ "learning_rate": 4.963621859363797e-05,
101
+ "loss": 0.6011715888977051,
102
+ "mean_token_accuracy": 0.8323964163661003,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6840810470802839,
109
+ "eval_loss": 0.6037028431892395,
110
+ "eval_mean_token_accuracy": 0.8309669033732525,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.4637,
113
+ "eval_samples_per_second": 15.914,
114
+ "eval_steps_per_second": 1.989,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6776216626167297,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.8918434977531433,
121
+ "learning_rate": 5.9663737501443624e-05,
122
+ "loss": 0.5991742610931396,
123
+ "mean_token_accuracy": 0.8300838828086853,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.690427724705186,
130
+ "eval_loss": 0.5939701795578003,
131
+ "eval_mean_token_accuracy": 0.8345950186945671,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.6626,
134
+ "eval_samples_per_second": 15.878,
135
+ "eval_steps_per_second": 1.985,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6709842771291733,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9135531187057495,
142
+ "learning_rate": 6.969125640924927e-05,
143
+ "loss": 0.5914147377014161,
144
+ "mean_token_accuracy": 0.8314545609056949,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6584504666023476,
151
+ "eval_loss": 0.5849721431732178,
152
+ "eval_mean_token_accuracy": 0.8357757236375365,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.3262,
155
+ "eval_samples_per_second": 15.94,
156
+ "eval_steps_per_second": 1.992,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.6524647936224938,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8651587963104248,
163
+ "learning_rate": 7.971877531705493e-05,
164
+ "loss": 0.5710843563079834,
165
+ "mean_token_accuracy": 0.8396127380430698,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6283470298661742,
172
+ "eval_loss": 0.5738973617553711,
173
+ "eval_mean_token_accuracy": 0.8379981181649274,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.5619,
176
+ "eval_samples_per_second": 15.896,
177
+ "eval_steps_per_second": 1.987,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6450445972383022,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8661723732948303,
184
+ "learning_rate": 8.974629422486058e-05,
185
+ "loss": 0.5677794933319091,
186
+ "mean_token_accuracy": 0.8389350369572639,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6142613257086554,
193
+ "eval_loss": 0.5698265433311462,
194
+ "eval_mean_token_accuracy": 0.8388577273418737,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.4443,
197
+ "eval_samples_per_second": 15.918,
198
+ "eval_steps_per_second": 1.99,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6448334597051144,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9662242531776428,
205
+ "learning_rate": 9.977381313266624e-05,
206
+ "loss": 0.581433916091919,
207
+ "mean_token_accuracy": 0.8387043006718159,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6154296522916749,
214
+ "eval_loss": 0.5660303831100464,
215
+ "eval_mean_token_accuracy": 0.8412494766850804,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3063,
218
+ "eval_samples_per_second": 15.943,
219
+ "eval_steps_per_second": 1.993,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6376728117465973,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7618638873100281,
226
+ "learning_rate": 0.00010980133204047189,
227
+ "loss": 0.5678351402282715,
228
+ "mean_token_accuracy": 0.8404546812176704,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6181817033956217,
235
+ "eval_loss": 0.5663750171661377,
236
+ "eval_mean_token_accuracy": 0.8388350962899452,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.5904,
239
+ "eval_samples_per_second": 15.891,
240
+ "eval_steps_per_second": 1.986,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6303176879882812,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.7571695446968079,
247
+ "learning_rate": 0.00011982885094827753,
248
+ "loss": 0.5502778053283691,
249
+ "mean_token_accuracy": 0.8429657347500324,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6252533817707107,
256
+ "eval_loss": 0.5570284128189087,
257
+ "eval_mean_token_accuracy": 0.8427327847064927,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.4157,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.99,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6202544964849949,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.6447190642356873,
268
+ "learning_rate": 0.00012985636985608318,
269
+ "loss": 0.5485352993011474,
270
+ "mean_token_accuracy": 0.844165726006031,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.6441633552312851,
277
+ "eval_loss": 0.5606644153594971,
278
+ "eval_mean_token_accuracy": 0.842403513054515,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.6343,
281
+ "eval_samples_per_second": 15.883,
282
+ "eval_steps_per_second": 1.985,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6306711677461863,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7869907021522522,
289
+ "learning_rate": 0.00013988388876388883,
290
+ "loss": 0.5579307556152344,
291
+ "mean_token_accuracy": 0.841247134655714,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.6263934678809587,
298
+ "eval_loss": 0.5559113025665283,
299
+ "eval_mean_token_accuracy": 0.8427334743183713,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.6403,
302
+ "eval_samples_per_second": 15.882,
303
+ "eval_steps_per_second": 1.985,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6385872110724449,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6679229736328125,
310
+ "learning_rate": 0.0001499114076716945,
311
+ "loss": 0.5667279720306396,
312
+ "mean_token_accuracy": 0.8389136254787445,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6141417321077612,
319
+ "eval_loss": 0.5570600628852844,
320
+ "eval_mean_token_accuracy": 0.8437647996253745,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.7588,
323
+ "eval_samples_per_second": 15.86,
324
+ "eval_steps_per_second": 1.983,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6199494235217571,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.7924400568008423,
331
+ "learning_rate": 0.00015993892657950015,
332
+ "loss": 0.5529299736022949,
333
+ "mean_token_accuracy": 0.8426973208785057,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6133768925833147,
340
+ "eval_loss": 0.556602418422699,
341
+ "eval_mean_token_accuracy": 0.8432947965555413,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.492,
344
+ "eval_samples_per_second": 15.909,
345
+ "eval_steps_per_second": 1.989,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6203986253589392,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8364354372024536,
352
+ "learning_rate": 0.00016996644548730578,
353
+ "loss": 0.5551144123077393,
354
+ "mean_token_accuracy": 0.8432973213493824,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6017442844634833,
361
+ "eval_loss": 0.5566568374633789,
362
+ "eval_mean_token_accuracy": 0.8437666123689607,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.5552,
365
+ "eval_samples_per_second": 15.897,
366
+ "eval_steps_per_second": 1.987,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6341533534228802,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.7783445715904236,
373
+ "learning_rate": 0.00017999396439511144,
374
+ "loss": 0.5669133186340332,
375
+ "mean_token_accuracy": 0.8379446342587471,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6055107958788095,
382
+ "eval_loss": 0.5599350333213806,
383
+ "eval_mean_token_accuracy": 0.8435030894917112,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.4814,
386
+ "eval_samples_per_second": 15.911,
387
+ "eval_steps_per_second": 1.989,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.6306198488920927,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.8449786901473999,
394
+ "learning_rate": 0.0001900214833029171,
395
+ "loss": 0.5739435195922852,
396
+ "mean_token_accuracy": 0.8393832489848136,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6129532439071078,
403
+ "eval_loss": 0.5566295981407166,
404
+ "eval_mean_token_accuracy": 0.8430350880290187,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.4643,
407
+ "eval_samples_per_second": 15.914,
408
+ "eval_steps_per_second": 1.989,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6203123550862074,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 0.7334314584732056,
415
+ "learning_rate": 0.00020004900221072276,
416
+ "loss": 0.5547565937042236,
417
+ "mean_token_accuracy": 0.8403573960065842,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6275761647279873,
424
+ "eval_loss": 0.5621116757392883,
425
+ "eval_mean_token_accuracy": 0.841587379228237,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.4748,
428
+ "eval_samples_per_second": 15.912,
429
+ "eval_steps_per_second": 1.989,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5795013002860241,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 0.8858296871185303,
436
+ "learning_rate": 0.0002015421505577756,
437
+ "loss": 0.5183939933776855,
438
+ "mean_token_accuracy": 0.850081592034071,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5583065545489622,
445
+ "eval_loss": 0.5605642199516296,
446
+ "eval_mean_token_accuracy": 0.8439708411000496,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.5422,
449
+ "eval_samples_per_second": 15.9,
450
+ "eval_steps_per_second": 1.987,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5671238023787737,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.6882498264312744,
457
+ "learning_rate": 0.00020150112347025443,
458
+ "loss": 0.5077326774597168,
459
+ "mean_token_accuracy": 0.8489868573844432,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5868900277933409,
466
+ "eval_loss": 0.5602695345878601,
467
+ "eval_mean_token_accuracy": 0.8428842161977014,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.623,
470
+ "eval_samples_per_second": 15.885,
471
+ "eval_steps_per_second": 1.986,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5533561781048775,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 0.7717723250389099,
478
+ "learning_rate": 0.0002014297192297181,
479
+ "loss": 0.4954517364501953,
480
+ "mean_token_accuracy": 0.8529035650193691,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5600803743961246,
487
+ "eval_loss": 0.5608077645301819,
488
+ "eval_mean_token_accuracy": 0.8445036771685578,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.1316,
491
+ "eval_samples_per_second": 15.976,
492
+ "eval_steps_per_second": 1.997,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5692154694348573,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 0.7322827577590942,
499
+ "learning_rate": 0.0002013279593707117,
500
+ "loss": 0.505049467086792,
501
+ "mean_token_accuracy": 0.8551576808094978,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5732695829383162,
508
+ "eval_loss": 0.5594323873519897,
509
+ "eval_mean_token_accuracy": 0.8449713407560836,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.2726,
512
+ "eval_samples_per_second": 15.949,
513
+ "eval_steps_per_second": 1.994,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5817618492990733,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 1.1776764392852783,
520
+ "learning_rate": 0.0002011958745826208,
521
+ "loss": 0.5137609958648681,
522
+ "mean_token_accuracy": 0.8521522544324398,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5662581343636957,
529
+ "eval_loss": 0.5595026016235352,
530
+ "eval_mean_token_accuracy": 0.8441977164773053,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7261,
533
+ "eval_samples_per_second": 15.866,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5712925456464291,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7960361838340759,
541
+ "learning_rate": 0.0002010335047004159,
542
+ "loss": 0.5134767532348633,
543
+ "mean_token_accuracy": 0.8513577707111836,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.5441222797299541,
550
+ "eval_loss": 0.5535460114479065,
551
+ "eval_mean_token_accuracy": 0.8450886118550633,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.2675,
554
+ "eval_samples_per_second": 15.95,
555
+ "eval_steps_per_second": 1.994,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5787045754492283,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9205410480499268,
562
+ "learning_rate": 0.00020084089869263887,
563
+ "loss": 0.5119701862335205,
564
+ "mean_token_accuracy": 0.8503516331315041,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.5744457827057949,
571
+ "eval_loss": 0.5514978766441345,
572
+ "eval_mean_token_accuracy": 0.845929987901865,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.2299,
575
+ "eval_samples_per_second": 15.957,
576
+ "eval_steps_per_second": 1.995,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5739392962306737,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.7475653886795044,
583
+ "learning_rate": 0.00020061811464663464,
584
+ "loss": 0.5189042091369629,
585
+ "mean_token_accuracy": 0.8492388024926185,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6116398271433142,
592
+ "eval_loss": 0.551732063293457,
593
+ "eval_mean_token_accuracy": 0.8450756967067719,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.6081,
596
+ "eval_samples_per_second": 15.888,
597
+ "eval_steps_per_second": 1.986,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5755622573196888,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.8218411803245544,
604
+ "learning_rate": 0.00020036521975103286,
605
+ "loss": 0.5106248378753662,
606
+ "mean_token_accuracy": 0.8506785586476326,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.5906928708386976,
613
+ "eval_loss": 0.551278829574585,
614
+ "eval_mean_token_accuracy": 0.8462819308042526,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.5438,
617
+ "eval_samples_per_second": 15.899,
618
+ "eval_steps_per_second": 1.987,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5694822132587433,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.8880652189254761,
625
+ "learning_rate": 0.00020008229027548475,
626
+ "loss": 0.5140334606170655,
627
+ "mean_token_accuracy": 0.8521522797644139,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5599641964532608,
634
+ "eval_loss": 0.5501875877380371,
635
+ "eval_mean_token_accuracy": 0.8467660788879838,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6458,
638
+ "eval_samples_per_second": 15.881,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.5675108034163714,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.837087094783783,
646
+ "learning_rate": 0.0001997694115476612,
647
+ "loss": 0.5099846363067627,
648
+ "mean_token_accuracy": 0.8543680295348167,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5728072581249614,
655
+ "eval_loss": 0.5445425510406494,
656
+ "eval_mean_token_accuracy": 0.8474342175001321,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.4859,
659
+ "eval_samples_per_second": 15.91,
660
+ "eval_steps_per_second": 1.989,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5700885068625212,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6598765850067139,
667
+ "learning_rate": 0.000199426677927519,
668
+ "loss": 0.5122694969177246,
669
+ "mean_token_accuracy": 0.8519927568733692,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5476993622128353,
676
+ "eval_loss": 0.5427973866462708,
677
+ "eval_mean_token_accuracy": 0.8478512147138285,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.4172,
680
+ "eval_samples_per_second": 15.923,
681
+ "eval_steps_per_second": 1.99,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.5829229176044464,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.6965194940567017,
688
+ "learning_rate": 0.00019905419277884342,
689
+ "loss": 0.5253659725189209,
690
+ "mean_token_accuracy": 0.8493309423327446,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5666290084983028,
697
+ "eval_loss": 0.5467478036880493,
698
+ "eval_mean_token_accuracy": 0.8479407703460649,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.4414,
701
+ "eval_samples_per_second": 15.918,
702
+ "eval_steps_per_second": 1.99,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5498311135917902,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.636583685874939,
709
+ "learning_rate": 0.00019865206843807482,
710
+ "loss": 0.49981012344360354,
711
+ "mean_token_accuracy": 0.8560848504304885,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.539117265406043,
718
+ "eval_loss": 0.53994220495224,
719
+ "eval_mean_token_accuracy": 0.8488582601380903,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.5296,
722
+ "eval_samples_per_second": 15.902,
723
+ "eval_steps_per_second": 1.988,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5543891470879316,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.6068442463874817,
730
+ "learning_rate": 0.0001982204261804297,
731
+ "loss": 0.498047399520874,
732
+ "mean_token_accuracy": 0.8554679051041603,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5703774151760478,
739
+ "eval_loss": 0.5300245881080627,
740
+ "eval_mean_token_accuracy": 0.850798153946566,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.6456,
743
+ "eval_samples_per_second": 15.881,
744
+ "eval_steps_per_second": 1.985,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.546524541825056,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7274155020713806,
751
+ "learning_rate": 0.00019775939618332566,
752
+ "loss": 0.4988589286804199,
753
+ "mean_token_accuracy": 0.853422473371029,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5614905688305234,
760
+ "eval_loss": 0.5350332260131836,
761
+ "eval_mean_token_accuracy": 0.8492204359797544,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 86.7581,
764
+ "eval_samples_per_second": 15.86,
765
+ "eval_steps_per_second": 1.983,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5519792139530182,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.663466215133667,
772
+ "learning_rate": 0.00019726911748712167,
773
+ "loss": 0.5099314212799072,
774
+ "mean_token_accuracy": 0.848412600159645,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5583519090053647,
781
+ "eval_loss": 0.530483603477478,
782
+ "eval_mean_token_accuracy": 0.8500003374593202,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 86.3961,
785
+ "eval_samples_per_second": 15.927,
786
+ "eval_steps_per_second": 1.991,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.5454779766499996,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.890394926071167,
793
+ "learning_rate": 0.00019674973795318548,
794
+ "loss": 0.4931994915008545,
795
+ "mean_token_accuracy": 0.8540832489728928,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.572755502406941,
802
+ "eval_loss": 0.5415747761726379,
803
+ "eval_mean_token_accuracy": 0.8444425803284312,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4323,
806
+ "eval_samples_per_second": 15.92,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.5392089951783419,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.632411777973175,
814
+ "learning_rate": 0.00019620141421930058,
815
+ "loss": 0.4957888603210449,
816
+ "mean_token_accuracy": 0.8549866065382957,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.540764772961306,
823
+ "eval_loss": 0.5327216386795044,
824
+ "eval_mean_token_accuracy": 0.850631088364956,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.8097,
827
+ "eval_samples_per_second": 15.851,
828
+ "eval_steps_per_second": 1.981,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5674678739160299,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6958843469619751,
835
+ "learning_rate": 0.0001956243116524263,
836
+ "loss": 0.504389762878418,
837
+ "mean_token_accuracy": 0.8527948908507824,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.530262403190136,
844
+ "eval_loss": 0.5308871865272522,
845
+ "eval_mean_token_accuracy": 0.8522498046242913,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.7942,
848
+ "eval_samples_per_second": 15.854,
849
+ "eval_steps_per_second": 1.982,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4742849511213792,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.6941492557525635,
856
+ "learning_rate": 0.00019501860429882556,
857
+ "loss": 0.418599271774292,
858
+ "mean_token_accuracy": 0.8748210859604371,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.504602165069691,
865
+ "eval_loss": 0.542878270149231,
866
+ "eval_mean_token_accuracy": 0.8507604484641275,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.7841,
869
+ "eval_samples_per_second": 15.855,
870
+ "eval_steps_per_second": 1.982,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.45857742577791216,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.5791997909545898,
877
+ "learning_rate": 0.00019438447483157478,
878
+ "loss": 0.399777889251709,
879
+ "mean_token_accuracy": 0.8754058346152306,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.5028848362176918,
886
+ "eval_loss": 0.5356478095054626,
887
+ "eval_mean_token_accuracy": 0.8525635412959165,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.6707,
890
+ "eval_samples_per_second": 15.876,
891
+ "eval_steps_per_second": 1.985,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4869446292519569,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6483516693115234,
898
+ "learning_rate": 0.00019372211449547223,
899
+ "loss": 0.40715818405151366,
900
+ "mean_token_accuracy": 0.875113020837307,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.4928991326759028,
907
+ "eval_loss": 0.5419561862945557,
908
+ "eval_mean_token_accuracy": 0.8516040146350861,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 87.0686,
911
+ "eval_samples_per_second": 15.804,
912
+ "eval_steps_per_second": 1.975,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.45819590501487256,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.6661920547485352,
919
+ "learning_rate": 0.00019303172304936108,
920
+ "loss": 0.39511430263519287,
921
+ "mean_token_accuracy": 0.8780680045485496,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.48602560647698334,
928
+ "eval_loss": 0.5436084866523743,
929
+ "eval_mean_token_accuracy": 0.8500938470973525,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.6809,
932
+ "eval_samples_per_second": 15.874,
933
+ "eval_steps_per_second": 1.984,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4780638810247183,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6870484352111816,
940
+ "learning_rate": 0.0001923135087058851,
941
+ "loss": 0.4061615467071533,
942
+ "mean_token_accuracy": 0.8766494184732437,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.48236206035281337,
949
+ "eval_loss": 0.5446090698242188,
950
+ "eval_mean_token_accuracy": 0.8507725513258646,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.7398,
953
+ "eval_samples_per_second": 15.864,
954
+ "eval_steps_per_second": 1.983,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.463029869645834,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.6894590854644775,
961
+ "learning_rate": 0.00019156768806869427,
962
+ "loss": 0.39602413177490237,
963
+ "mean_token_accuracy": 0.876420046389103,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.4904779093556626,
970
+ "eval_loss": 0.5404934287071228,
971
+ "eval_mean_token_accuracy": 0.852238280828609,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.5348,
974
+ "eval_samples_per_second": 15.901,
975
+ "eval_steps_per_second": 1.988,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4817025110125542,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7756227254867554,
982
+ "learning_rate": 0.00019079448606712033,
983
+ "loss": 0.4177968502044678,
984
+ "mean_token_accuracy": 0.8712256088852882,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5153802815218305,
991
+ "eval_loss": 0.5424937605857849,
992
+ "eval_mean_token_accuracy": 0.8506565759348315,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.8973,
995
+ "eval_samples_per_second": 15.835,
996
+ "eval_steps_per_second": 1.979,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.46456389091908934,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 1.2000319957733154,
1003
+ "learning_rate": 0.00018999413588834105,
1004
+ "loss": 0.4084665775299072,
1005
+ "mean_token_accuracy": 0.8750658087432385,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4849439303195754,
1012
+ "eval_loss": 0.545662522315979,
1013
+ "eval_mean_token_accuracy": 0.8491013112456299,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.9049,
1016
+ "eval_samples_per_second": 15.833,
1017
+ "eval_steps_per_second": 1.979,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.4857471022754908,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.9696341753005981,
1024
+ "learning_rate": 0.0001891668789070541,
1025
+ "loss": 0.4149796962738037,
1026
+ "mean_token_accuracy": 0.8704176343977451,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.4872790058684904,
1033
+ "eval_loss": 0.5412707924842834,
1034
+ "eval_mean_token_accuracy": 0.8509329602468846,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.7846,
1037
+ "eval_samples_per_second": 15.855,
1038
+ "eval_steps_per_second": 1.982,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.4727417893707752,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.7852500677108765,
1045
+ "learning_rate": 0.0001883129646126818,
1046
+ "loss": 0.4142886161804199,
1047
+ "mean_token_accuracy": 0.8712429471313954,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5386548059624295,
1054
+ "eval_loss": 0.536101222038269,
1055
+ "eval_mean_token_accuracy": 0.8499491239009902,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.9501,
1058
+ "eval_samples_per_second": 15.825,
1059
+ "eval_steps_per_second": 1.978,
1060
+ "step": 1000
1061
+ }
1062
+ ],
1063
+ "logging_steps": 20,
1064
+ "max_steps": 4020,
1065
+ "num_input_tokens_seen": 0,
1066
+ "num_train_epochs": 10,
1067
+ "save_steps": 20,
1068
+ "stateful_callbacks": {
1069
+ "TrainerControl": {
1070
+ "args": {
1071
+ "should_epoch_stop": false,
1072
+ "should_evaluate": false,
1073
+ "should_log": false,
1074
+ "should_save": true,
1075
+ "should_training_stop": false
1076
+ },
1077
+ "attributes": {}
1078
+ }
1079
+ },
1080
+ "total_flos": 9.859037950771814e+16,
1081
+ "train_batch_size": 4,
1082
+ "trial_name": null,
1083
+ "trial_params": null
1084
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1020/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1020/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0005183818805460705,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1020/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1020/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1020/trainer_state.json ADDED
@@ -0,0 +1,1105 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.5379825653798256,
6
+ "eval_steps": 20,
7
+ "global_step": 1020,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.9784346982836722,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.0229668617248535,
16
+ "learning_rate": 9.526142962415369e-06,
17
+ "loss": 1.7360023498535155,
18
+ "mean_token_accuracy": 0.6449888605624438,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.41506897571475,
25
+ "eval_loss": 1.1876318454742432,
26
+ "eval_mean_token_accuracy": 0.734131895525511,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.8071,
29
+ "eval_samples_per_second": 15.671,
30
+ "eval_steps_per_second": 1.959,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.049924298375845,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.5795097351074219,
37
+ "learning_rate": 1.9553661870221022e-05,
38
+ "loss": 0.8944448471069336,
39
+ "mean_token_accuracy": 0.7748479396104813,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7996658658565476,
46
+ "eval_loss": 0.7202735543251038,
47
+ "eval_mean_token_accuracy": 0.8070558306089667,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.9199,
50
+ "eval_samples_per_second": 15.831,
51
+ "eval_steps_per_second": 1.979,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7734908878803253,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3136248588562012,
58
+ "learning_rate": 2.9581180778026673e-05,
59
+ "loss": 0.6780608654022217,
60
+ "mean_token_accuracy": 0.8168170280754566,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7119324009778888,
67
+ "eval_loss": 0.6554311513900757,
68
+ "eval_mean_token_accuracy": 0.8215604798738346,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.8692,
71
+ "eval_samples_per_second": 15.84,
72
+ "eval_steps_per_second": 1.98,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7071127541363239,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.387060284614563,
79
+ "learning_rate": 3.960869968583232e-05,
80
+ "loss": 0.6382100582122803,
81
+ "mean_token_accuracy": 0.8229366384446621,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6883931482254073,
88
+ "eval_loss": 0.625065803527832,
89
+ "eval_mean_token_accuracy": 0.828940509710201,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.662,
92
+ "eval_samples_per_second": 15.878,
93
+ "eval_steps_per_second": 1.985,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6800824083387852,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9892916679382324,
100
+ "learning_rate": 4.963621859363797e-05,
101
+ "loss": 0.6011715888977051,
102
+ "mean_token_accuracy": 0.8323964163661003,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6840810470802839,
109
+ "eval_loss": 0.6037028431892395,
110
+ "eval_mean_token_accuracy": 0.8309669033732525,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.4637,
113
+ "eval_samples_per_second": 15.914,
114
+ "eval_steps_per_second": 1.989,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6776216626167297,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.8918434977531433,
121
+ "learning_rate": 5.9663737501443624e-05,
122
+ "loss": 0.5991742610931396,
123
+ "mean_token_accuracy": 0.8300838828086853,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.690427724705186,
130
+ "eval_loss": 0.5939701795578003,
131
+ "eval_mean_token_accuracy": 0.8345950186945671,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.6626,
134
+ "eval_samples_per_second": 15.878,
135
+ "eval_steps_per_second": 1.985,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6709842771291733,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9135531187057495,
142
+ "learning_rate": 6.969125640924927e-05,
143
+ "loss": 0.5914147377014161,
144
+ "mean_token_accuracy": 0.8314545609056949,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6584504666023476,
151
+ "eval_loss": 0.5849721431732178,
152
+ "eval_mean_token_accuracy": 0.8357757236375365,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.3262,
155
+ "eval_samples_per_second": 15.94,
156
+ "eval_steps_per_second": 1.992,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.6524647936224938,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8651587963104248,
163
+ "learning_rate": 7.971877531705493e-05,
164
+ "loss": 0.5710843563079834,
165
+ "mean_token_accuracy": 0.8396127380430698,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6283470298661742,
172
+ "eval_loss": 0.5738973617553711,
173
+ "eval_mean_token_accuracy": 0.8379981181649274,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.5619,
176
+ "eval_samples_per_second": 15.896,
177
+ "eval_steps_per_second": 1.987,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6450445972383022,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8661723732948303,
184
+ "learning_rate": 8.974629422486058e-05,
185
+ "loss": 0.5677794933319091,
186
+ "mean_token_accuracy": 0.8389350369572639,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6142613257086554,
193
+ "eval_loss": 0.5698265433311462,
194
+ "eval_mean_token_accuracy": 0.8388577273418737,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.4443,
197
+ "eval_samples_per_second": 15.918,
198
+ "eval_steps_per_second": 1.99,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6448334597051144,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9662242531776428,
205
+ "learning_rate": 9.977381313266624e-05,
206
+ "loss": 0.581433916091919,
207
+ "mean_token_accuracy": 0.8387043006718159,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6154296522916749,
214
+ "eval_loss": 0.5660303831100464,
215
+ "eval_mean_token_accuracy": 0.8412494766850804,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3063,
218
+ "eval_samples_per_second": 15.943,
219
+ "eval_steps_per_second": 1.993,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6376728117465973,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7618638873100281,
226
+ "learning_rate": 0.00010980133204047189,
227
+ "loss": 0.5678351402282715,
228
+ "mean_token_accuracy": 0.8404546812176704,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6181817033956217,
235
+ "eval_loss": 0.5663750171661377,
236
+ "eval_mean_token_accuracy": 0.8388350962899452,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.5904,
239
+ "eval_samples_per_second": 15.891,
240
+ "eval_steps_per_second": 1.986,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6303176879882812,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.7571695446968079,
247
+ "learning_rate": 0.00011982885094827753,
248
+ "loss": 0.5502778053283691,
249
+ "mean_token_accuracy": 0.8429657347500324,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6252533817707107,
256
+ "eval_loss": 0.5570284128189087,
257
+ "eval_mean_token_accuracy": 0.8427327847064927,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.4157,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.99,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6202544964849949,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.6447190642356873,
268
+ "learning_rate": 0.00012985636985608318,
269
+ "loss": 0.5485352993011474,
270
+ "mean_token_accuracy": 0.844165726006031,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.6441633552312851,
277
+ "eval_loss": 0.5606644153594971,
278
+ "eval_mean_token_accuracy": 0.842403513054515,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.6343,
281
+ "eval_samples_per_second": 15.883,
282
+ "eval_steps_per_second": 1.985,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6306711677461863,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7869907021522522,
289
+ "learning_rate": 0.00013988388876388883,
290
+ "loss": 0.5579307556152344,
291
+ "mean_token_accuracy": 0.841247134655714,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.6263934678809587,
298
+ "eval_loss": 0.5559113025665283,
299
+ "eval_mean_token_accuracy": 0.8427334743183713,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.6403,
302
+ "eval_samples_per_second": 15.882,
303
+ "eval_steps_per_second": 1.985,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6385872110724449,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6679229736328125,
310
+ "learning_rate": 0.0001499114076716945,
311
+ "loss": 0.5667279720306396,
312
+ "mean_token_accuracy": 0.8389136254787445,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6141417321077612,
319
+ "eval_loss": 0.5570600628852844,
320
+ "eval_mean_token_accuracy": 0.8437647996253745,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.7588,
323
+ "eval_samples_per_second": 15.86,
324
+ "eval_steps_per_second": 1.983,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6199494235217571,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.7924400568008423,
331
+ "learning_rate": 0.00015993892657950015,
332
+ "loss": 0.5529299736022949,
333
+ "mean_token_accuracy": 0.8426973208785057,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6133768925833147,
340
+ "eval_loss": 0.556602418422699,
341
+ "eval_mean_token_accuracy": 0.8432947965555413,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.492,
344
+ "eval_samples_per_second": 15.909,
345
+ "eval_steps_per_second": 1.989,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6203986253589392,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8364354372024536,
352
+ "learning_rate": 0.00016996644548730578,
353
+ "loss": 0.5551144123077393,
354
+ "mean_token_accuracy": 0.8432973213493824,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6017442844634833,
361
+ "eval_loss": 0.5566568374633789,
362
+ "eval_mean_token_accuracy": 0.8437666123689607,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.5552,
365
+ "eval_samples_per_second": 15.897,
366
+ "eval_steps_per_second": 1.987,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6341533534228802,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.7783445715904236,
373
+ "learning_rate": 0.00017999396439511144,
374
+ "loss": 0.5669133186340332,
375
+ "mean_token_accuracy": 0.8379446342587471,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6055107958788095,
382
+ "eval_loss": 0.5599350333213806,
383
+ "eval_mean_token_accuracy": 0.8435030894917112,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.4814,
386
+ "eval_samples_per_second": 15.911,
387
+ "eval_steps_per_second": 1.989,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.6306198488920927,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.8449786901473999,
394
+ "learning_rate": 0.0001900214833029171,
395
+ "loss": 0.5739435195922852,
396
+ "mean_token_accuracy": 0.8393832489848136,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6129532439071078,
403
+ "eval_loss": 0.5566295981407166,
404
+ "eval_mean_token_accuracy": 0.8430350880290187,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.4643,
407
+ "eval_samples_per_second": 15.914,
408
+ "eval_steps_per_second": 1.989,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6203123550862074,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 0.7334314584732056,
415
+ "learning_rate": 0.00020004900221072276,
416
+ "loss": 0.5547565937042236,
417
+ "mean_token_accuracy": 0.8403573960065842,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6275761647279873,
424
+ "eval_loss": 0.5621116757392883,
425
+ "eval_mean_token_accuracy": 0.841587379228237,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.4748,
428
+ "eval_samples_per_second": 15.912,
429
+ "eval_steps_per_second": 1.989,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5795013002860241,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 0.8858296871185303,
436
+ "learning_rate": 0.0002015421505577756,
437
+ "loss": 0.5183939933776855,
438
+ "mean_token_accuracy": 0.850081592034071,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5583065545489622,
445
+ "eval_loss": 0.5605642199516296,
446
+ "eval_mean_token_accuracy": 0.8439708411000496,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.5422,
449
+ "eval_samples_per_second": 15.9,
450
+ "eval_steps_per_second": 1.987,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5671238023787737,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.6882498264312744,
457
+ "learning_rate": 0.00020150112347025443,
458
+ "loss": 0.5077326774597168,
459
+ "mean_token_accuracy": 0.8489868573844432,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5868900277933409,
466
+ "eval_loss": 0.5602695345878601,
467
+ "eval_mean_token_accuracy": 0.8428842161977014,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.623,
470
+ "eval_samples_per_second": 15.885,
471
+ "eval_steps_per_second": 1.986,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5533561781048775,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 0.7717723250389099,
478
+ "learning_rate": 0.0002014297192297181,
479
+ "loss": 0.4954517364501953,
480
+ "mean_token_accuracy": 0.8529035650193691,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5600803743961246,
487
+ "eval_loss": 0.5608077645301819,
488
+ "eval_mean_token_accuracy": 0.8445036771685578,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.1316,
491
+ "eval_samples_per_second": 15.976,
492
+ "eval_steps_per_second": 1.997,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5692154694348573,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 0.7322827577590942,
499
+ "learning_rate": 0.0002013279593707117,
500
+ "loss": 0.505049467086792,
501
+ "mean_token_accuracy": 0.8551576808094978,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5732695829383162,
508
+ "eval_loss": 0.5594323873519897,
509
+ "eval_mean_token_accuracy": 0.8449713407560836,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.2726,
512
+ "eval_samples_per_second": 15.949,
513
+ "eval_steps_per_second": 1.994,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5817618492990733,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 1.1776764392852783,
520
+ "learning_rate": 0.0002011958745826208,
521
+ "loss": 0.5137609958648681,
522
+ "mean_token_accuracy": 0.8521522544324398,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5662581343636957,
529
+ "eval_loss": 0.5595026016235352,
530
+ "eval_mean_token_accuracy": 0.8441977164773053,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7261,
533
+ "eval_samples_per_second": 15.866,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5712925456464291,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7960361838340759,
541
+ "learning_rate": 0.0002010335047004159,
542
+ "loss": 0.5134767532348633,
543
+ "mean_token_accuracy": 0.8513577707111836,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.5441222797299541,
550
+ "eval_loss": 0.5535460114479065,
551
+ "eval_mean_token_accuracy": 0.8450886118550633,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.2675,
554
+ "eval_samples_per_second": 15.95,
555
+ "eval_steps_per_second": 1.994,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5787045754492283,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9205410480499268,
562
+ "learning_rate": 0.00020084089869263887,
563
+ "loss": 0.5119701862335205,
564
+ "mean_token_accuracy": 0.8503516331315041,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.5744457827057949,
571
+ "eval_loss": 0.5514978766441345,
572
+ "eval_mean_token_accuracy": 0.845929987901865,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.2299,
575
+ "eval_samples_per_second": 15.957,
576
+ "eval_steps_per_second": 1.995,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5739392962306737,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.7475653886795044,
583
+ "learning_rate": 0.00020061811464663464,
584
+ "loss": 0.5189042091369629,
585
+ "mean_token_accuracy": 0.8492388024926185,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6116398271433142,
592
+ "eval_loss": 0.551732063293457,
593
+ "eval_mean_token_accuracy": 0.8450756967067719,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.6081,
596
+ "eval_samples_per_second": 15.888,
597
+ "eval_steps_per_second": 1.986,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5755622573196888,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.8218411803245544,
604
+ "learning_rate": 0.00020036521975103286,
605
+ "loss": 0.5106248378753662,
606
+ "mean_token_accuracy": 0.8506785586476326,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.5906928708386976,
613
+ "eval_loss": 0.551278829574585,
614
+ "eval_mean_token_accuracy": 0.8462819308042526,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.5438,
617
+ "eval_samples_per_second": 15.899,
618
+ "eval_steps_per_second": 1.987,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5694822132587433,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.8880652189254761,
625
+ "learning_rate": 0.00020008229027548475,
626
+ "loss": 0.5140334606170655,
627
+ "mean_token_accuracy": 0.8521522797644139,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5599641964532608,
634
+ "eval_loss": 0.5501875877380371,
635
+ "eval_mean_token_accuracy": 0.8467660788879838,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6458,
638
+ "eval_samples_per_second": 15.881,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.5675108034163714,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.837087094783783,
646
+ "learning_rate": 0.0001997694115476612,
647
+ "loss": 0.5099846363067627,
648
+ "mean_token_accuracy": 0.8543680295348167,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5728072581249614,
655
+ "eval_loss": 0.5445425510406494,
656
+ "eval_mean_token_accuracy": 0.8474342175001321,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.4859,
659
+ "eval_samples_per_second": 15.91,
660
+ "eval_steps_per_second": 1.989,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5700885068625212,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6598765850067139,
667
+ "learning_rate": 0.000199426677927519,
668
+ "loss": 0.5122694969177246,
669
+ "mean_token_accuracy": 0.8519927568733692,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5476993622128353,
676
+ "eval_loss": 0.5427973866462708,
677
+ "eval_mean_token_accuracy": 0.8478512147138285,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.4172,
680
+ "eval_samples_per_second": 15.923,
681
+ "eval_steps_per_second": 1.99,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.5829229176044464,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.6965194940567017,
688
+ "learning_rate": 0.00019905419277884342,
689
+ "loss": 0.5253659725189209,
690
+ "mean_token_accuracy": 0.8493309423327446,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5666290084983028,
697
+ "eval_loss": 0.5467478036880493,
698
+ "eval_mean_token_accuracy": 0.8479407703460649,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.4414,
701
+ "eval_samples_per_second": 15.918,
702
+ "eval_steps_per_second": 1.99,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5498311135917902,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.636583685874939,
709
+ "learning_rate": 0.00019865206843807482,
710
+ "loss": 0.49981012344360354,
711
+ "mean_token_accuracy": 0.8560848504304885,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.539117265406043,
718
+ "eval_loss": 0.53994220495224,
719
+ "eval_mean_token_accuracy": 0.8488582601380903,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.5296,
722
+ "eval_samples_per_second": 15.902,
723
+ "eval_steps_per_second": 1.988,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5543891470879316,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.6068442463874817,
730
+ "learning_rate": 0.0001982204261804297,
731
+ "loss": 0.498047399520874,
732
+ "mean_token_accuracy": 0.8554679051041603,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5703774151760478,
739
+ "eval_loss": 0.5300245881080627,
740
+ "eval_mean_token_accuracy": 0.850798153946566,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.6456,
743
+ "eval_samples_per_second": 15.881,
744
+ "eval_steps_per_second": 1.985,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.546524541825056,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7274155020713806,
751
+ "learning_rate": 0.00019775939618332566,
752
+ "loss": 0.4988589286804199,
753
+ "mean_token_accuracy": 0.853422473371029,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5614905688305234,
760
+ "eval_loss": 0.5350332260131836,
761
+ "eval_mean_token_accuracy": 0.8492204359797544,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 86.7581,
764
+ "eval_samples_per_second": 15.86,
765
+ "eval_steps_per_second": 1.983,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5519792139530182,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.663466215133667,
772
+ "learning_rate": 0.00019726911748712167,
773
+ "loss": 0.5099314212799072,
774
+ "mean_token_accuracy": 0.848412600159645,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5583519090053647,
781
+ "eval_loss": 0.530483603477478,
782
+ "eval_mean_token_accuracy": 0.8500003374593202,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 86.3961,
785
+ "eval_samples_per_second": 15.927,
786
+ "eval_steps_per_second": 1.991,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.5454779766499996,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.890394926071167,
793
+ "learning_rate": 0.00019674973795318548,
794
+ "loss": 0.4931994915008545,
795
+ "mean_token_accuracy": 0.8540832489728928,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.572755502406941,
802
+ "eval_loss": 0.5415747761726379,
803
+ "eval_mean_token_accuracy": 0.8444425803284312,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4323,
806
+ "eval_samples_per_second": 15.92,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.5392089951783419,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.632411777973175,
814
+ "learning_rate": 0.00019620141421930058,
815
+ "loss": 0.4957888603210449,
816
+ "mean_token_accuracy": 0.8549866065382957,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.540764772961306,
823
+ "eval_loss": 0.5327216386795044,
824
+ "eval_mean_token_accuracy": 0.850631088364956,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.8097,
827
+ "eval_samples_per_second": 15.851,
828
+ "eval_steps_per_second": 1.981,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5674678739160299,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6958843469619751,
835
+ "learning_rate": 0.0001956243116524263,
836
+ "loss": 0.504389762878418,
837
+ "mean_token_accuracy": 0.8527948908507824,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.530262403190136,
844
+ "eval_loss": 0.5308871865272522,
845
+ "eval_mean_token_accuracy": 0.8522498046242913,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.7942,
848
+ "eval_samples_per_second": 15.854,
849
+ "eval_steps_per_second": 1.982,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4742849511213792,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.6941492557525635,
856
+ "learning_rate": 0.00019501860429882556,
857
+ "loss": 0.418599271774292,
858
+ "mean_token_accuracy": 0.8748210859604371,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.504602165069691,
865
+ "eval_loss": 0.542878270149231,
866
+ "eval_mean_token_accuracy": 0.8507604484641275,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.7841,
869
+ "eval_samples_per_second": 15.855,
870
+ "eval_steps_per_second": 1.982,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.45857742577791216,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.5791997909545898,
877
+ "learning_rate": 0.00019438447483157478,
878
+ "loss": 0.399777889251709,
879
+ "mean_token_accuracy": 0.8754058346152306,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.5028848362176918,
886
+ "eval_loss": 0.5356478095054626,
887
+ "eval_mean_token_accuracy": 0.8525635412959165,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.6707,
890
+ "eval_samples_per_second": 15.876,
891
+ "eval_steps_per_second": 1.985,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4869446292519569,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6483516693115234,
898
+ "learning_rate": 0.00019372211449547223,
899
+ "loss": 0.40715818405151366,
900
+ "mean_token_accuracy": 0.875113020837307,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.4928991326759028,
907
+ "eval_loss": 0.5419561862945557,
908
+ "eval_mean_token_accuracy": 0.8516040146350861,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 87.0686,
911
+ "eval_samples_per_second": 15.804,
912
+ "eval_steps_per_second": 1.975,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.45819590501487256,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.6661920547485352,
919
+ "learning_rate": 0.00019303172304936108,
920
+ "loss": 0.39511430263519287,
921
+ "mean_token_accuracy": 0.8780680045485496,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.48602560647698334,
928
+ "eval_loss": 0.5436084866523743,
929
+ "eval_mean_token_accuracy": 0.8500938470973525,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.6809,
932
+ "eval_samples_per_second": 15.874,
933
+ "eval_steps_per_second": 1.984,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4780638810247183,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6870484352111816,
940
+ "learning_rate": 0.0001923135087058851,
941
+ "loss": 0.4061615467071533,
942
+ "mean_token_accuracy": 0.8766494184732437,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.48236206035281337,
949
+ "eval_loss": 0.5446090698242188,
950
+ "eval_mean_token_accuracy": 0.8507725513258646,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.7398,
953
+ "eval_samples_per_second": 15.864,
954
+ "eval_steps_per_second": 1.983,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.463029869645834,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.6894590854644775,
961
+ "learning_rate": 0.00019156768806869427,
962
+ "loss": 0.39602413177490237,
963
+ "mean_token_accuracy": 0.876420046389103,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.4904779093556626,
970
+ "eval_loss": 0.5404934287071228,
971
+ "eval_mean_token_accuracy": 0.852238280828609,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.5348,
974
+ "eval_samples_per_second": 15.901,
975
+ "eval_steps_per_second": 1.988,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4817025110125542,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7756227254867554,
982
+ "learning_rate": 0.00019079448606712033,
983
+ "loss": 0.4177968502044678,
984
+ "mean_token_accuracy": 0.8712256088852882,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5153802815218305,
991
+ "eval_loss": 0.5424937605857849,
992
+ "eval_mean_token_accuracy": 0.8506565759348315,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.8973,
995
+ "eval_samples_per_second": 15.835,
996
+ "eval_steps_per_second": 1.979,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.46456389091908934,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 1.2000319957733154,
1003
+ "learning_rate": 0.00018999413588834105,
1004
+ "loss": 0.4084665775299072,
1005
+ "mean_token_accuracy": 0.8750658087432385,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4849439303195754,
1012
+ "eval_loss": 0.545662522315979,
1013
+ "eval_mean_token_accuracy": 0.8491013112456299,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.9049,
1016
+ "eval_samples_per_second": 15.833,
1017
+ "eval_steps_per_second": 1.979,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.4857471022754908,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.9696341753005981,
1024
+ "learning_rate": 0.0001891668789070541,
1025
+ "loss": 0.4149796962738037,
1026
+ "mean_token_accuracy": 0.8704176343977451,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.4872790058684904,
1033
+ "eval_loss": 0.5412707924842834,
1034
+ "eval_mean_token_accuracy": 0.8509329602468846,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.7846,
1037
+ "eval_samples_per_second": 15.855,
1038
+ "eval_steps_per_second": 1.982,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.4727417893707752,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.7852500677108765,
1045
+ "learning_rate": 0.0001883129646126818,
1046
+ "loss": 0.4142886161804199,
1047
+ "mean_token_accuracy": 0.8712429471313954,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5386548059624295,
1054
+ "eval_loss": 0.536101222038269,
1055
+ "eval_mean_token_accuracy": 0.8499491239009902,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.9501,
1058
+ "eval_samples_per_second": 15.825,
1059
+ "eval_steps_per_second": 1.978,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.4673406321555376,
1064
+ "epoch": 2.5379825653798256,
1065
+ "grad_norm": 0.7133921384811401,
1066
+ "learning_rate": 0.0001874326505341286,
1067
+ "loss": 0.40857529640197754,
1068
+ "mean_token_accuracy": 0.8747925907373428,
1069
+ "num_tokens": 2384270.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.5379825653798256,
1074
+ "eval_entropy": 0.495788364909416,
1075
+ "eval_loss": 0.5418923497200012,
1076
+ "eval_mean_token_accuracy": 0.851321972040243,
1077
+ "eval_num_tokens": 2384270.0,
1078
+ "eval_runtime": 86.7154,
1079
+ "eval_samples_per_second": 15.868,
1080
+ "eval_steps_per_second": 1.983,
1081
+ "step": 1020
1082
+ }
1083
+ ],
1084
+ "logging_steps": 20,
1085
+ "max_steps": 4020,
1086
+ "num_input_tokens_seen": 0,
1087
+ "num_train_epochs": 10,
1088
+ "save_steps": 20,
1089
+ "stateful_callbacks": {
1090
+ "TrainerControl": {
1091
+ "args": {
1092
+ "should_epoch_stop": false,
1093
+ "should_evaluate": false,
1094
+ "should_log": false,
1095
+ "should_save": true,
1096
+ "should_training_stop": false
1097
+ },
1098
+ "attributes": {}
1099
+ }
1100
+ },
1101
+ "total_flos": 1.0076952699436032e+17,
1102
+ "train_batch_size": 4,
1103
+ "trial_name": null,
1104
+ "trial_params": null
1105
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1040/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1040/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0005183818805460705,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1040/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1040/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1040/trainer_state.json ADDED
@@ -0,0 +1,1126 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.587795765877958,
6
+ "eval_steps": 20,
7
+ "global_step": 1040,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.9784346982836722,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.0229668617248535,
16
+ "learning_rate": 9.526142962415369e-06,
17
+ "loss": 1.7360023498535155,
18
+ "mean_token_accuracy": 0.6449888605624438,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.41506897571475,
25
+ "eval_loss": 1.1876318454742432,
26
+ "eval_mean_token_accuracy": 0.734131895525511,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.8071,
29
+ "eval_samples_per_second": 15.671,
30
+ "eval_steps_per_second": 1.959,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.049924298375845,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.5795097351074219,
37
+ "learning_rate": 1.9553661870221022e-05,
38
+ "loss": 0.8944448471069336,
39
+ "mean_token_accuracy": 0.7748479396104813,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7996658658565476,
46
+ "eval_loss": 0.7202735543251038,
47
+ "eval_mean_token_accuracy": 0.8070558306089667,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.9199,
50
+ "eval_samples_per_second": 15.831,
51
+ "eval_steps_per_second": 1.979,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7734908878803253,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3136248588562012,
58
+ "learning_rate": 2.9581180778026673e-05,
59
+ "loss": 0.6780608654022217,
60
+ "mean_token_accuracy": 0.8168170280754566,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7119324009778888,
67
+ "eval_loss": 0.6554311513900757,
68
+ "eval_mean_token_accuracy": 0.8215604798738346,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.8692,
71
+ "eval_samples_per_second": 15.84,
72
+ "eval_steps_per_second": 1.98,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7071127541363239,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.387060284614563,
79
+ "learning_rate": 3.960869968583232e-05,
80
+ "loss": 0.6382100582122803,
81
+ "mean_token_accuracy": 0.8229366384446621,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6883931482254073,
88
+ "eval_loss": 0.625065803527832,
89
+ "eval_mean_token_accuracy": 0.828940509710201,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.662,
92
+ "eval_samples_per_second": 15.878,
93
+ "eval_steps_per_second": 1.985,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6800824083387852,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9892916679382324,
100
+ "learning_rate": 4.963621859363797e-05,
101
+ "loss": 0.6011715888977051,
102
+ "mean_token_accuracy": 0.8323964163661003,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6840810470802839,
109
+ "eval_loss": 0.6037028431892395,
110
+ "eval_mean_token_accuracy": 0.8309669033732525,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.4637,
113
+ "eval_samples_per_second": 15.914,
114
+ "eval_steps_per_second": 1.989,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6776216626167297,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.8918434977531433,
121
+ "learning_rate": 5.9663737501443624e-05,
122
+ "loss": 0.5991742610931396,
123
+ "mean_token_accuracy": 0.8300838828086853,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.690427724705186,
130
+ "eval_loss": 0.5939701795578003,
131
+ "eval_mean_token_accuracy": 0.8345950186945671,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.6626,
134
+ "eval_samples_per_second": 15.878,
135
+ "eval_steps_per_second": 1.985,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6709842771291733,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9135531187057495,
142
+ "learning_rate": 6.969125640924927e-05,
143
+ "loss": 0.5914147377014161,
144
+ "mean_token_accuracy": 0.8314545609056949,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6584504666023476,
151
+ "eval_loss": 0.5849721431732178,
152
+ "eval_mean_token_accuracy": 0.8357757236375365,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.3262,
155
+ "eval_samples_per_second": 15.94,
156
+ "eval_steps_per_second": 1.992,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.6524647936224938,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8651587963104248,
163
+ "learning_rate": 7.971877531705493e-05,
164
+ "loss": 0.5710843563079834,
165
+ "mean_token_accuracy": 0.8396127380430698,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6283470298661742,
172
+ "eval_loss": 0.5738973617553711,
173
+ "eval_mean_token_accuracy": 0.8379981181649274,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.5619,
176
+ "eval_samples_per_second": 15.896,
177
+ "eval_steps_per_second": 1.987,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6450445972383022,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8661723732948303,
184
+ "learning_rate": 8.974629422486058e-05,
185
+ "loss": 0.5677794933319091,
186
+ "mean_token_accuracy": 0.8389350369572639,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6142613257086554,
193
+ "eval_loss": 0.5698265433311462,
194
+ "eval_mean_token_accuracy": 0.8388577273418737,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.4443,
197
+ "eval_samples_per_second": 15.918,
198
+ "eval_steps_per_second": 1.99,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6448334597051144,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9662242531776428,
205
+ "learning_rate": 9.977381313266624e-05,
206
+ "loss": 0.581433916091919,
207
+ "mean_token_accuracy": 0.8387043006718159,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6154296522916749,
214
+ "eval_loss": 0.5660303831100464,
215
+ "eval_mean_token_accuracy": 0.8412494766850804,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3063,
218
+ "eval_samples_per_second": 15.943,
219
+ "eval_steps_per_second": 1.993,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6376728117465973,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7618638873100281,
226
+ "learning_rate": 0.00010980133204047189,
227
+ "loss": 0.5678351402282715,
228
+ "mean_token_accuracy": 0.8404546812176704,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6181817033956217,
235
+ "eval_loss": 0.5663750171661377,
236
+ "eval_mean_token_accuracy": 0.8388350962899452,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.5904,
239
+ "eval_samples_per_second": 15.891,
240
+ "eval_steps_per_second": 1.986,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6303176879882812,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.7571695446968079,
247
+ "learning_rate": 0.00011982885094827753,
248
+ "loss": 0.5502778053283691,
249
+ "mean_token_accuracy": 0.8429657347500324,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6252533817707107,
256
+ "eval_loss": 0.5570284128189087,
257
+ "eval_mean_token_accuracy": 0.8427327847064927,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.4157,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.99,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6202544964849949,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.6447190642356873,
268
+ "learning_rate": 0.00012985636985608318,
269
+ "loss": 0.5485352993011474,
270
+ "mean_token_accuracy": 0.844165726006031,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.6441633552312851,
277
+ "eval_loss": 0.5606644153594971,
278
+ "eval_mean_token_accuracy": 0.842403513054515,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.6343,
281
+ "eval_samples_per_second": 15.883,
282
+ "eval_steps_per_second": 1.985,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6306711677461863,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7869907021522522,
289
+ "learning_rate": 0.00013988388876388883,
290
+ "loss": 0.5579307556152344,
291
+ "mean_token_accuracy": 0.841247134655714,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.6263934678809587,
298
+ "eval_loss": 0.5559113025665283,
299
+ "eval_mean_token_accuracy": 0.8427334743183713,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.6403,
302
+ "eval_samples_per_second": 15.882,
303
+ "eval_steps_per_second": 1.985,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6385872110724449,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6679229736328125,
310
+ "learning_rate": 0.0001499114076716945,
311
+ "loss": 0.5667279720306396,
312
+ "mean_token_accuracy": 0.8389136254787445,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6141417321077612,
319
+ "eval_loss": 0.5570600628852844,
320
+ "eval_mean_token_accuracy": 0.8437647996253745,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.7588,
323
+ "eval_samples_per_second": 15.86,
324
+ "eval_steps_per_second": 1.983,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6199494235217571,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.7924400568008423,
331
+ "learning_rate": 0.00015993892657950015,
332
+ "loss": 0.5529299736022949,
333
+ "mean_token_accuracy": 0.8426973208785057,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6133768925833147,
340
+ "eval_loss": 0.556602418422699,
341
+ "eval_mean_token_accuracy": 0.8432947965555413,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.492,
344
+ "eval_samples_per_second": 15.909,
345
+ "eval_steps_per_second": 1.989,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6203986253589392,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8364354372024536,
352
+ "learning_rate": 0.00016996644548730578,
353
+ "loss": 0.5551144123077393,
354
+ "mean_token_accuracy": 0.8432973213493824,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6017442844634833,
361
+ "eval_loss": 0.5566568374633789,
362
+ "eval_mean_token_accuracy": 0.8437666123689607,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.5552,
365
+ "eval_samples_per_second": 15.897,
366
+ "eval_steps_per_second": 1.987,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6341533534228802,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.7783445715904236,
373
+ "learning_rate": 0.00017999396439511144,
374
+ "loss": 0.5669133186340332,
375
+ "mean_token_accuracy": 0.8379446342587471,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6055107958788095,
382
+ "eval_loss": 0.5599350333213806,
383
+ "eval_mean_token_accuracy": 0.8435030894917112,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.4814,
386
+ "eval_samples_per_second": 15.911,
387
+ "eval_steps_per_second": 1.989,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.6306198488920927,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.8449786901473999,
394
+ "learning_rate": 0.0001900214833029171,
395
+ "loss": 0.5739435195922852,
396
+ "mean_token_accuracy": 0.8393832489848136,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6129532439071078,
403
+ "eval_loss": 0.5566295981407166,
404
+ "eval_mean_token_accuracy": 0.8430350880290187,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.4643,
407
+ "eval_samples_per_second": 15.914,
408
+ "eval_steps_per_second": 1.989,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6203123550862074,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 0.7334314584732056,
415
+ "learning_rate": 0.00020004900221072276,
416
+ "loss": 0.5547565937042236,
417
+ "mean_token_accuracy": 0.8403573960065842,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6275761647279873,
424
+ "eval_loss": 0.5621116757392883,
425
+ "eval_mean_token_accuracy": 0.841587379228237,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.4748,
428
+ "eval_samples_per_second": 15.912,
429
+ "eval_steps_per_second": 1.989,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5795013002860241,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 0.8858296871185303,
436
+ "learning_rate": 0.0002015421505577756,
437
+ "loss": 0.5183939933776855,
438
+ "mean_token_accuracy": 0.850081592034071,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5583065545489622,
445
+ "eval_loss": 0.5605642199516296,
446
+ "eval_mean_token_accuracy": 0.8439708411000496,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.5422,
449
+ "eval_samples_per_second": 15.9,
450
+ "eval_steps_per_second": 1.987,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5671238023787737,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.6882498264312744,
457
+ "learning_rate": 0.00020150112347025443,
458
+ "loss": 0.5077326774597168,
459
+ "mean_token_accuracy": 0.8489868573844432,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5868900277933409,
466
+ "eval_loss": 0.5602695345878601,
467
+ "eval_mean_token_accuracy": 0.8428842161977014,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.623,
470
+ "eval_samples_per_second": 15.885,
471
+ "eval_steps_per_second": 1.986,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5533561781048775,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 0.7717723250389099,
478
+ "learning_rate": 0.0002014297192297181,
479
+ "loss": 0.4954517364501953,
480
+ "mean_token_accuracy": 0.8529035650193691,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5600803743961246,
487
+ "eval_loss": 0.5608077645301819,
488
+ "eval_mean_token_accuracy": 0.8445036771685578,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.1316,
491
+ "eval_samples_per_second": 15.976,
492
+ "eval_steps_per_second": 1.997,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5692154694348573,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 0.7322827577590942,
499
+ "learning_rate": 0.0002013279593707117,
500
+ "loss": 0.505049467086792,
501
+ "mean_token_accuracy": 0.8551576808094978,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5732695829383162,
508
+ "eval_loss": 0.5594323873519897,
509
+ "eval_mean_token_accuracy": 0.8449713407560836,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.2726,
512
+ "eval_samples_per_second": 15.949,
513
+ "eval_steps_per_second": 1.994,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5817618492990733,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 1.1776764392852783,
520
+ "learning_rate": 0.0002011958745826208,
521
+ "loss": 0.5137609958648681,
522
+ "mean_token_accuracy": 0.8521522544324398,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5662581343636957,
529
+ "eval_loss": 0.5595026016235352,
530
+ "eval_mean_token_accuracy": 0.8441977164773053,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7261,
533
+ "eval_samples_per_second": 15.866,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5712925456464291,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7960361838340759,
541
+ "learning_rate": 0.0002010335047004159,
542
+ "loss": 0.5134767532348633,
543
+ "mean_token_accuracy": 0.8513577707111836,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.5441222797299541,
550
+ "eval_loss": 0.5535460114479065,
551
+ "eval_mean_token_accuracy": 0.8450886118550633,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.2675,
554
+ "eval_samples_per_second": 15.95,
555
+ "eval_steps_per_second": 1.994,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5787045754492283,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9205410480499268,
562
+ "learning_rate": 0.00020084089869263887,
563
+ "loss": 0.5119701862335205,
564
+ "mean_token_accuracy": 0.8503516331315041,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.5744457827057949,
571
+ "eval_loss": 0.5514978766441345,
572
+ "eval_mean_token_accuracy": 0.845929987901865,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.2299,
575
+ "eval_samples_per_second": 15.957,
576
+ "eval_steps_per_second": 1.995,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5739392962306737,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.7475653886795044,
583
+ "learning_rate": 0.00020061811464663464,
584
+ "loss": 0.5189042091369629,
585
+ "mean_token_accuracy": 0.8492388024926185,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6116398271433142,
592
+ "eval_loss": 0.551732063293457,
593
+ "eval_mean_token_accuracy": 0.8450756967067719,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.6081,
596
+ "eval_samples_per_second": 15.888,
597
+ "eval_steps_per_second": 1.986,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5755622573196888,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.8218411803245544,
604
+ "learning_rate": 0.00020036521975103286,
605
+ "loss": 0.5106248378753662,
606
+ "mean_token_accuracy": 0.8506785586476326,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.5906928708386976,
613
+ "eval_loss": 0.551278829574585,
614
+ "eval_mean_token_accuracy": 0.8462819308042526,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.5438,
617
+ "eval_samples_per_second": 15.899,
618
+ "eval_steps_per_second": 1.987,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5694822132587433,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.8880652189254761,
625
+ "learning_rate": 0.00020008229027548475,
626
+ "loss": 0.5140334606170655,
627
+ "mean_token_accuracy": 0.8521522797644139,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5599641964532608,
634
+ "eval_loss": 0.5501875877380371,
635
+ "eval_mean_token_accuracy": 0.8467660788879838,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6458,
638
+ "eval_samples_per_second": 15.881,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.5675108034163714,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.837087094783783,
646
+ "learning_rate": 0.0001997694115476612,
647
+ "loss": 0.5099846363067627,
648
+ "mean_token_accuracy": 0.8543680295348167,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5728072581249614,
655
+ "eval_loss": 0.5445425510406494,
656
+ "eval_mean_token_accuracy": 0.8474342175001321,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.4859,
659
+ "eval_samples_per_second": 15.91,
660
+ "eval_steps_per_second": 1.989,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5700885068625212,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6598765850067139,
667
+ "learning_rate": 0.000199426677927519,
668
+ "loss": 0.5122694969177246,
669
+ "mean_token_accuracy": 0.8519927568733692,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5476993622128353,
676
+ "eval_loss": 0.5427973866462708,
677
+ "eval_mean_token_accuracy": 0.8478512147138285,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.4172,
680
+ "eval_samples_per_second": 15.923,
681
+ "eval_steps_per_second": 1.99,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.5829229176044464,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.6965194940567017,
688
+ "learning_rate": 0.00019905419277884342,
689
+ "loss": 0.5253659725189209,
690
+ "mean_token_accuracy": 0.8493309423327446,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5666290084983028,
697
+ "eval_loss": 0.5467478036880493,
698
+ "eval_mean_token_accuracy": 0.8479407703460649,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.4414,
701
+ "eval_samples_per_second": 15.918,
702
+ "eval_steps_per_second": 1.99,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5498311135917902,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.636583685874939,
709
+ "learning_rate": 0.00019865206843807482,
710
+ "loss": 0.49981012344360354,
711
+ "mean_token_accuracy": 0.8560848504304885,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.539117265406043,
718
+ "eval_loss": 0.53994220495224,
719
+ "eval_mean_token_accuracy": 0.8488582601380903,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.5296,
722
+ "eval_samples_per_second": 15.902,
723
+ "eval_steps_per_second": 1.988,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5543891470879316,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.6068442463874817,
730
+ "learning_rate": 0.0001982204261804297,
731
+ "loss": 0.498047399520874,
732
+ "mean_token_accuracy": 0.8554679051041603,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5703774151760478,
739
+ "eval_loss": 0.5300245881080627,
740
+ "eval_mean_token_accuracy": 0.850798153946566,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.6456,
743
+ "eval_samples_per_second": 15.881,
744
+ "eval_steps_per_second": 1.985,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.546524541825056,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7274155020713806,
751
+ "learning_rate": 0.00019775939618332566,
752
+ "loss": 0.4988589286804199,
753
+ "mean_token_accuracy": 0.853422473371029,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5614905688305234,
760
+ "eval_loss": 0.5350332260131836,
761
+ "eval_mean_token_accuracy": 0.8492204359797544,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 86.7581,
764
+ "eval_samples_per_second": 15.86,
765
+ "eval_steps_per_second": 1.983,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5519792139530182,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.663466215133667,
772
+ "learning_rate": 0.00019726911748712167,
773
+ "loss": 0.5099314212799072,
774
+ "mean_token_accuracy": 0.848412600159645,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5583519090053647,
781
+ "eval_loss": 0.530483603477478,
782
+ "eval_mean_token_accuracy": 0.8500003374593202,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 86.3961,
785
+ "eval_samples_per_second": 15.927,
786
+ "eval_steps_per_second": 1.991,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.5454779766499996,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.890394926071167,
793
+ "learning_rate": 0.00019674973795318548,
794
+ "loss": 0.4931994915008545,
795
+ "mean_token_accuracy": 0.8540832489728928,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.572755502406941,
802
+ "eval_loss": 0.5415747761726379,
803
+ "eval_mean_token_accuracy": 0.8444425803284312,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4323,
806
+ "eval_samples_per_second": 15.92,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.5392089951783419,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.632411777973175,
814
+ "learning_rate": 0.00019620141421930058,
815
+ "loss": 0.4957888603210449,
816
+ "mean_token_accuracy": 0.8549866065382957,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.540764772961306,
823
+ "eval_loss": 0.5327216386795044,
824
+ "eval_mean_token_accuracy": 0.850631088364956,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.8097,
827
+ "eval_samples_per_second": 15.851,
828
+ "eval_steps_per_second": 1.981,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5674678739160299,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6958843469619751,
835
+ "learning_rate": 0.0001956243116524263,
836
+ "loss": 0.504389762878418,
837
+ "mean_token_accuracy": 0.8527948908507824,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.530262403190136,
844
+ "eval_loss": 0.5308871865272522,
845
+ "eval_mean_token_accuracy": 0.8522498046242913,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.7942,
848
+ "eval_samples_per_second": 15.854,
849
+ "eval_steps_per_second": 1.982,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4742849511213792,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.6941492557525635,
856
+ "learning_rate": 0.00019501860429882556,
857
+ "loss": 0.418599271774292,
858
+ "mean_token_accuracy": 0.8748210859604371,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.504602165069691,
865
+ "eval_loss": 0.542878270149231,
866
+ "eval_mean_token_accuracy": 0.8507604484641275,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.7841,
869
+ "eval_samples_per_second": 15.855,
870
+ "eval_steps_per_second": 1.982,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.45857742577791216,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.5791997909545898,
877
+ "learning_rate": 0.00019438447483157478,
878
+ "loss": 0.399777889251709,
879
+ "mean_token_accuracy": 0.8754058346152306,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.5028848362176918,
886
+ "eval_loss": 0.5356478095054626,
887
+ "eval_mean_token_accuracy": 0.8525635412959165,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.6707,
890
+ "eval_samples_per_second": 15.876,
891
+ "eval_steps_per_second": 1.985,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4869446292519569,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6483516693115234,
898
+ "learning_rate": 0.00019372211449547223,
899
+ "loss": 0.40715818405151366,
900
+ "mean_token_accuracy": 0.875113020837307,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.4928991326759028,
907
+ "eval_loss": 0.5419561862945557,
908
+ "eval_mean_token_accuracy": 0.8516040146350861,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 87.0686,
911
+ "eval_samples_per_second": 15.804,
912
+ "eval_steps_per_second": 1.975,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.45819590501487256,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.6661920547485352,
919
+ "learning_rate": 0.00019303172304936108,
920
+ "loss": 0.39511430263519287,
921
+ "mean_token_accuracy": 0.8780680045485496,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.48602560647698334,
928
+ "eval_loss": 0.5436084866523743,
929
+ "eval_mean_token_accuracy": 0.8500938470973525,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.6809,
932
+ "eval_samples_per_second": 15.874,
933
+ "eval_steps_per_second": 1.984,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4780638810247183,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6870484352111816,
940
+ "learning_rate": 0.0001923135087058851,
941
+ "loss": 0.4061615467071533,
942
+ "mean_token_accuracy": 0.8766494184732437,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.48236206035281337,
949
+ "eval_loss": 0.5446090698242188,
950
+ "eval_mean_token_accuracy": 0.8507725513258646,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.7398,
953
+ "eval_samples_per_second": 15.864,
954
+ "eval_steps_per_second": 1.983,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.463029869645834,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.6894590854644775,
961
+ "learning_rate": 0.00019156768806869427,
962
+ "loss": 0.39602413177490237,
963
+ "mean_token_accuracy": 0.876420046389103,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.4904779093556626,
970
+ "eval_loss": 0.5404934287071228,
971
+ "eval_mean_token_accuracy": 0.852238280828609,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.5348,
974
+ "eval_samples_per_second": 15.901,
975
+ "eval_steps_per_second": 1.988,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4817025110125542,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7756227254867554,
982
+ "learning_rate": 0.00019079448606712033,
983
+ "loss": 0.4177968502044678,
984
+ "mean_token_accuracy": 0.8712256088852882,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5153802815218305,
991
+ "eval_loss": 0.5424937605857849,
992
+ "eval_mean_token_accuracy": 0.8506565759348315,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.8973,
995
+ "eval_samples_per_second": 15.835,
996
+ "eval_steps_per_second": 1.979,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.46456389091908934,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 1.2000319957733154,
1003
+ "learning_rate": 0.00018999413588834105,
1004
+ "loss": 0.4084665775299072,
1005
+ "mean_token_accuracy": 0.8750658087432385,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4849439303195754,
1012
+ "eval_loss": 0.545662522315979,
1013
+ "eval_mean_token_accuracy": 0.8491013112456299,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.9049,
1016
+ "eval_samples_per_second": 15.833,
1017
+ "eval_steps_per_second": 1.979,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.4857471022754908,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.9696341753005981,
1024
+ "learning_rate": 0.0001891668789070541,
1025
+ "loss": 0.4149796962738037,
1026
+ "mean_token_accuracy": 0.8704176343977451,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.4872790058684904,
1033
+ "eval_loss": 0.5412707924842834,
1034
+ "eval_mean_token_accuracy": 0.8509329602468846,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.7846,
1037
+ "eval_samples_per_second": 15.855,
1038
+ "eval_steps_per_second": 1.982,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.4727417893707752,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.7852500677108765,
1045
+ "learning_rate": 0.0001883129646126818,
1046
+ "loss": 0.4142886161804199,
1047
+ "mean_token_accuracy": 0.8712429471313954,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5386548059624295,
1054
+ "eval_loss": 0.536101222038269,
1055
+ "eval_mean_token_accuracy": 0.8499491239009902,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.9501,
1058
+ "eval_samples_per_second": 15.825,
1059
+ "eval_steps_per_second": 1.978,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.4673406321555376,
1064
+ "epoch": 2.5379825653798256,
1065
+ "grad_norm": 0.7133921384811401,
1066
+ "learning_rate": 0.0001874326505341286,
1067
+ "loss": 0.40857529640197754,
1068
+ "mean_token_accuracy": 0.8747925907373428,
1069
+ "num_tokens": 2384270.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.5379825653798256,
1074
+ "eval_entropy": 0.495788364909416,
1075
+ "eval_loss": 0.5418923497200012,
1076
+ "eval_mean_token_accuracy": 0.851321972040243,
1077
+ "eval_num_tokens": 2384270.0,
1078
+ "eval_runtime": 86.7154,
1079
+ "eval_samples_per_second": 15.868,
1080
+ "eval_steps_per_second": 1.983,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.47599745728075504,
1085
+ "epoch": 2.587795765877958,
1086
+ "grad_norm": 0.8202953338623047,
1087
+ "learning_rate": 0.0001865262021621137,
1088
+ "loss": 0.40998234748840334,
1089
+ "mean_token_accuracy": 0.8758242674171924,
1090
+ "num_tokens": 2428036.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.587795765877958,
1095
+ "eval_entropy": 0.4887966953737791,
1096
+ "eval_loss": 0.5408804416656494,
1097
+ "eval_mean_token_accuracy": 0.8512661065473113,
1098
+ "eval_num_tokens": 2428036.0,
1099
+ "eval_runtime": 86.7869,
1100
+ "eval_samples_per_second": 15.855,
1101
+ "eval_steps_per_second": 1.982,
1102
+ "step": 1040
1103
+ }
1104
+ ],
1105
+ "logging_steps": 20,
1106
+ "max_steps": 4020,
1107
+ "num_input_tokens_seen": 0,
1108
+ "num_train_epochs": 10,
1109
+ "save_steps": 20,
1110
+ "stateful_callbacks": {
1111
+ "TrainerControl": {
1112
+ "args": {
1113
+ "should_epoch_stop": false,
1114
+ "should_evaluate": false,
1115
+ "should_log": false,
1116
+ "should_save": true,
1117
+ "should_training_stop": false
1118
+ },
1119
+ "attributes": {}
1120
+ }
1121
+ },
1122
+ "total_flos": 1.0254458345271091e+17,
1123
+ "train_batch_size": 4,
1124
+ "trial_name": null,
1125
+ "trial_params": null
1126
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1060/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1060/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0005183818805460705,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1060/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1060/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1060/trainer_state.json ADDED
@@ -0,0 +1,1147 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.6376089663760895,
6
+ "eval_steps": 20,
7
+ "global_step": 1060,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.9784346982836722,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.0229668617248535,
16
+ "learning_rate": 9.526142962415369e-06,
17
+ "loss": 1.7360023498535155,
18
+ "mean_token_accuracy": 0.6449888605624438,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.41506897571475,
25
+ "eval_loss": 1.1876318454742432,
26
+ "eval_mean_token_accuracy": 0.734131895525511,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.8071,
29
+ "eval_samples_per_second": 15.671,
30
+ "eval_steps_per_second": 1.959,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.049924298375845,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.5795097351074219,
37
+ "learning_rate": 1.9553661870221022e-05,
38
+ "loss": 0.8944448471069336,
39
+ "mean_token_accuracy": 0.7748479396104813,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7996658658565476,
46
+ "eval_loss": 0.7202735543251038,
47
+ "eval_mean_token_accuracy": 0.8070558306089667,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.9199,
50
+ "eval_samples_per_second": 15.831,
51
+ "eval_steps_per_second": 1.979,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7734908878803253,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3136248588562012,
58
+ "learning_rate": 2.9581180778026673e-05,
59
+ "loss": 0.6780608654022217,
60
+ "mean_token_accuracy": 0.8168170280754566,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7119324009778888,
67
+ "eval_loss": 0.6554311513900757,
68
+ "eval_mean_token_accuracy": 0.8215604798738346,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.8692,
71
+ "eval_samples_per_second": 15.84,
72
+ "eval_steps_per_second": 1.98,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7071127541363239,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.387060284614563,
79
+ "learning_rate": 3.960869968583232e-05,
80
+ "loss": 0.6382100582122803,
81
+ "mean_token_accuracy": 0.8229366384446621,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6883931482254073,
88
+ "eval_loss": 0.625065803527832,
89
+ "eval_mean_token_accuracy": 0.828940509710201,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.662,
92
+ "eval_samples_per_second": 15.878,
93
+ "eval_steps_per_second": 1.985,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6800824083387852,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9892916679382324,
100
+ "learning_rate": 4.963621859363797e-05,
101
+ "loss": 0.6011715888977051,
102
+ "mean_token_accuracy": 0.8323964163661003,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6840810470802839,
109
+ "eval_loss": 0.6037028431892395,
110
+ "eval_mean_token_accuracy": 0.8309669033732525,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.4637,
113
+ "eval_samples_per_second": 15.914,
114
+ "eval_steps_per_second": 1.989,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6776216626167297,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.8918434977531433,
121
+ "learning_rate": 5.9663737501443624e-05,
122
+ "loss": 0.5991742610931396,
123
+ "mean_token_accuracy": 0.8300838828086853,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.690427724705186,
130
+ "eval_loss": 0.5939701795578003,
131
+ "eval_mean_token_accuracy": 0.8345950186945671,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.6626,
134
+ "eval_samples_per_second": 15.878,
135
+ "eval_steps_per_second": 1.985,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6709842771291733,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9135531187057495,
142
+ "learning_rate": 6.969125640924927e-05,
143
+ "loss": 0.5914147377014161,
144
+ "mean_token_accuracy": 0.8314545609056949,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6584504666023476,
151
+ "eval_loss": 0.5849721431732178,
152
+ "eval_mean_token_accuracy": 0.8357757236375365,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.3262,
155
+ "eval_samples_per_second": 15.94,
156
+ "eval_steps_per_second": 1.992,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.6524647936224938,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8651587963104248,
163
+ "learning_rate": 7.971877531705493e-05,
164
+ "loss": 0.5710843563079834,
165
+ "mean_token_accuracy": 0.8396127380430698,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6283470298661742,
172
+ "eval_loss": 0.5738973617553711,
173
+ "eval_mean_token_accuracy": 0.8379981181649274,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.5619,
176
+ "eval_samples_per_second": 15.896,
177
+ "eval_steps_per_second": 1.987,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6450445972383022,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8661723732948303,
184
+ "learning_rate": 8.974629422486058e-05,
185
+ "loss": 0.5677794933319091,
186
+ "mean_token_accuracy": 0.8389350369572639,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6142613257086554,
193
+ "eval_loss": 0.5698265433311462,
194
+ "eval_mean_token_accuracy": 0.8388577273418737,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.4443,
197
+ "eval_samples_per_second": 15.918,
198
+ "eval_steps_per_second": 1.99,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6448334597051144,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9662242531776428,
205
+ "learning_rate": 9.977381313266624e-05,
206
+ "loss": 0.581433916091919,
207
+ "mean_token_accuracy": 0.8387043006718159,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6154296522916749,
214
+ "eval_loss": 0.5660303831100464,
215
+ "eval_mean_token_accuracy": 0.8412494766850804,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3063,
218
+ "eval_samples_per_second": 15.943,
219
+ "eval_steps_per_second": 1.993,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6376728117465973,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7618638873100281,
226
+ "learning_rate": 0.00010980133204047189,
227
+ "loss": 0.5678351402282715,
228
+ "mean_token_accuracy": 0.8404546812176704,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6181817033956217,
235
+ "eval_loss": 0.5663750171661377,
236
+ "eval_mean_token_accuracy": 0.8388350962899452,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.5904,
239
+ "eval_samples_per_second": 15.891,
240
+ "eval_steps_per_second": 1.986,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6303176879882812,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.7571695446968079,
247
+ "learning_rate": 0.00011982885094827753,
248
+ "loss": 0.5502778053283691,
249
+ "mean_token_accuracy": 0.8429657347500324,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6252533817707107,
256
+ "eval_loss": 0.5570284128189087,
257
+ "eval_mean_token_accuracy": 0.8427327847064927,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.4157,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.99,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6202544964849949,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.6447190642356873,
268
+ "learning_rate": 0.00012985636985608318,
269
+ "loss": 0.5485352993011474,
270
+ "mean_token_accuracy": 0.844165726006031,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.6441633552312851,
277
+ "eval_loss": 0.5606644153594971,
278
+ "eval_mean_token_accuracy": 0.842403513054515,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.6343,
281
+ "eval_samples_per_second": 15.883,
282
+ "eval_steps_per_second": 1.985,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6306711677461863,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7869907021522522,
289
+ "learning_rate": 0.00013988388876388883,
290
+ "loss": 0.5579307556152344,
291
+ "mean_token_accuracy": 0.841247134655714,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.6263934678809587,
298
+ "eval_loss": 0.5559113025665283,
299
+ "eval_mean_token_accuracy": 0.8427334743183713,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.6403,
302
+ "eval_samples_per_second": 15.882,
303
+ "eval_steps_per_second": 1.985,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6385872110724449,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6679229736328125,
310
+ "learning_rate": 0.0001499114076716945,
311
+ "loss": 0.5667279720306396,
312
+ "mean_token_accuracy": 0.8389136254787445,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6141417321077612,
319
+ "eval_loss": 0.5570600628852844,
320
+ "eval_mean_token_accuracy": 0.8437647996253745,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.7588,
323
+ "eval_samples_per_second": 15.86,
324
+ "eval_steps_per_second": 1.983,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6199494235217571,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.7924400568008423,
331
+ "learning_rate": 0.00015993892657950015,
332
+ "loss": 0.5529299736022949,
333
+ "mean_token_accuracy": 0.8426973208785057,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6133768925833147,
340
+ "eval_loss": 0.556602418422699,
341
+ "eval_mean_token_accuracy": 0.8432947965555413,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.492,
344
+ "eval_samples_per_second": 15.909,
345
+ "eval_steps_per_second": 1.989,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6203986253589392,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8364354372024536,
352
+ "learning_rate": 0.00016996644548730578,
353
+ "loss": 0.5551144123077393,
354
+ "mean_token_accuracy": 0.8432973213493824,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6017442844634833,
361
+ "eval_loss": 0.5566568374633789,
362
+ "eval_mean_token_accuracy": 0.8437666123689607,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.5552,
365
+ "eval_samples_per_second": 15.897,
366
+ "eval_steps_per_second": 1.987,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6341533534228802,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.7783445715904236,
373
+ "learning_rate": 0.00017999396439511144,
374
+ "loss": 0.5669133186340332,
375
+ "mean_token_accuracy": 0.8379446342587471,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6055107958788095,
382
+ "eval_loss": 0.5599350333213806,
383
+ "eval_mean_token_accuracy": 0.8435030894917112,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.4814,
386
+ "eval_samples_per_second": 15.911,
387
+ "eval_steps_per_second": 1.989,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.6306198488920927,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.8449786901473999,
394
+ "learning_rate": 0.0001900214833029171,
395
+ "loss": 0.5739435195922852,
396
+ "mean_token_accuracy": 0.8393832489848136,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6129532439071078,
403
+ "eval_loss": 0.5566295981407166,
404
+ "eval_mean_token_accuracy": 0.8430350880290187,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.4643,
407
+ "eval_samples_per_second": 15.914,
408
+ "eval_steps_per_second": 1.989,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6203123550862074,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 0.7334314584732056,
415
+ "learning_rate": 0.00020004900221072276,
416
+ "loss": 0.5547565937042236,
417
+ "mean_token_accuracy": 0.8403573960065842,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6275761647279873,
424
+ "eval_loss": 0.5621116757392883,
425
+ "eval_mean_token_accuracy": 0.841587379228237,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.4748,
428
+ "eval_samples_per_second": 15.912,
429
+ "eval_steps_per_second": 1.989,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5795013002860241,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 0.8858296871185303,
436
+ "learning_rate": 0.0002015421505577756,
437
+ "loss": 0.5183939933776855,
438
+ "mean_token_accuracy": 0.850081592034071,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5583065545489622,
445
+ "eval_loss": 0.5605642199516296,
446
+ "eval_mean_token_accuracy": 0.8439708411000496,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.5422,
449
+ "eval_samples_per_second": 15.9,
450
+ "eval_steps_per_second": 1.987,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5671238023787737,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.6882498264312744,
457
+ "learning_rate": 0.00020150112347025443,
458
+ "loss": 0.5077326774597168,
459
+ "mean_token_accuracy": 0.8489868573844432,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5868900277933409,
466
+ "eval_loss": 0.5602695345878601,
467
+ "eval_mean_token_accuracy": 0.8428842161977014,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.623,
470
+ "eval_samples_per_second": 15.885,
471
+ "eval_steps_per_second": 1.986,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5533561781048775,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 0.7717723250389099,
478
+ "learning_rate": 0.0002014297192297181,
479
+ "loss": 0.4954517364501953,
480
+ "mean_token_accuracy": 0.8529035650193691,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5600803743961246,
487
+ "eval_loss": 0.5608077645301819,
488
+ "eval_mean_token_accuracy": 0.8445036771685578,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.1316,
491
+ "eval_samples_per_second": 15.976,
492
+ "eval_steps_per_second": 1.997,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5692154694348573,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 0.7322827577590942,
499
+ "learning_rate": 0.0002013279593707117,
500
+ "loss": 0.505049467086792,
501
+ "mean_token_accuracy": 0.8551576808094978,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5732695829383162,
508
+ "eval_loss": 0.5594323873519897,
509
+ "eval_mean_token_accuracy": 0.8449713407560836,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.2726,
512
+ "eval_samples_per_second": 15.949,
513
+ "eval_steps_per_second": 1.994,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5817618492990733,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 1.1776764392852783,
520
+ "learning_rate": 0.0002011958745826208,
521
+ "loss": 0.5137609958648681,
522
+ "mean_token_accuracy": 0.8521522544324398,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5662581343636957,
529
+ "eval_loss": 0.5595026016235352,
530
+ "eval_mean_token_accuracy": 0.8441977164773053,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7261,
533
+ "eval_samples_per_second": 15.866,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5712925456464291,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7960361838340759,
541
+ "learning_rate": 0.0002010335047004159,
542
+ "loss": 0.5134767532348633,
543
+ "mean_token_accuracy": 0.8513577707111836,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.5441222797299541,
550
+ "eval_loss": 0.5535460114479065,
551
+ "eval_mean_token_accuracy": 0.8450886118550633,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.2675,
554
+ "eval_samples_per_second": 15.95,
555
+ "eval_steps_per_second": 1.994,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5787045754492283,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9205410480499268,
562
+ "learning_rate": 0.00020084089869263887,
563
+ "loss": 0.5119701862335205,
564
+ "mean_token_accuracy": 0.8503516331315041,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.5744457827057949,
571
+ "eval_loss": 0.5514978766441345,
572
+ "eval_mean_token_accuracy": 0.845929987901865,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.2299,
575
+ "eval_samples_per_second": 15.957,
576
+ "eval_steps_per_second": 1.995,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5739392962306737,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.7475653886795044,
583
+ "learning_rate": 0.00020061811464663464,
584
+ "loss": 0.5189042091369629,
585
+ "mean_token_accuracy": 0.8492388024926185,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6116398271433142,
592
+ "eval_loss": 0.551732063293457,
593
+ "eval_mean_token_accuracy": 0.8450756967067719,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.6081,
596
+ "eval_samples_per_second": 15.888,
597
+ "eval_steps_per_second": 1.986,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5755622573196888,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.8218411803245544,
604
+ "learning_rate": 0.00020036521975103286,
605
+ "loss": 0.5106248378753662,
606
+ "mean_token_accuracy": 0.8506785586476326,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.5906928708386976,
613
+ "eval_loss": 0.551278829574585,
614
+ "eval_mean_token_accuracy": 0.8462819308042526,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.5438,
617
+ "eval_samples_per_second": 15.899,
618
+ "eval_steps_per_second": 1.987,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5694822132587433,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.8880652189254761,
625
+ "learning_rate": 0.00020008229027548475,
626
+ "loss": 0.5140334606170655,
627
+ "mean_token_accuracy": 0.8521522797644139,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5599641964532608,
634
+ "eval_loss": 0.5501875877380371,
635
+ "eval_mean_token_accuracy": 0.8467660788879838,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6458,
638
+ "eval_samples_per_second": 15.881,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.5675108034163714,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.837087094783783,
646
+ "learning_rate": 0.0001997694115476612,
647
+ "loss": 0.5099846363067627,
648
+ "mean_token_accuracy": 0.8543680295348167,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5728072581249614,
655
+ "eval_loss": 0.5445425510406494,
656
+ "eval_mean_token_accuracy": 0.8474342175001321,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.4859,
659
+ "eval_samples_per_second": 15.91,
660
+ "eval_steps_per_second": 1.989,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5700885068625212,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6598765850067139,
667
+ "learning_rate": 0.000199426677927519,
668
+ "loss": 0.5122694969177246,
669
+ "mean_token_accuracy": 0.8519927568733692,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5476993622128353,
676
+ "eval_loss": 0.5427973866462708,
677
+ "eval_mean_token_accuracy": 0.8478512147138285,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.4172,
680
+ "eval_samples_per_second": 15.923,
681
+ "eval_steps_per_second": 1.99,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.5829229176044464,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.6965194940567017,
688
+ "learning_rate": 0.00019905419277884342,
689
+ "loss": 0.5253659725189209,
690
+ "mean_token_accuracy": 0.8493309423327446,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5666290084983028,
697
+ "eval_loss": 0.5467478036880493,
698
+ "eval_mean_token_accuracy": 0.8479407703460649,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.4414,
701
+ "eval_samples_per_second": 15.918,
702
+ "eval_steps_per_second": 1.99,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5498311135917902,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.636583685874939,
709
+ "learning_rate": 0.00019865206843807482,
710
+ "loss": 0.49981012344360354,
711
+ "mean_token_accuracy": 0.8560848504304885,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.539117265406043,
718
+ "eval_loss": 0.53994220495224,
719
+ "eval_mean_token_accuracy": 0.8488582601380903,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.5296,
722
+ "eval_samples_per_second": 15.902,
723
+ "eval_steps_per_second": 1.988,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5543891470879316,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.6068442463874817,
730
+ "learning_rate": 0.0001982204261804297,
731
+ "loss": 0.498047399520874,
732
+ "mean_token_accuracy": 0.8554679051041603,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5703774151760478,
739
+ "eval_loss": 0.5300245881080627,
740
+ "eval_mean_token_accuracy": 0.850798153946566,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.6456,
743
+ "eval_samples_per_second": 15.881,
744
+ "eval_steps_per_second": 1.985,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.546524541825056,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7274155020713806,
751
+ "learning_rate": 0.00019775939618332566,
752
+ "loss": 0.4988589286804199,
753
+ "mean_token_accuracy": 0.853422473371029,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5614905688305234,
760
+ "eval_loss": 0.5350332260131836,
761
+ "eval_mean_token_accuracy": 0.8492204359797544,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 86.7581,
764
+ "eval_samples_per_second": 15.86,
765
+ "eval_steps_per_second": 1.983,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5519792139530182,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.663466215133667,
772
+ "learning_rate": 0.00019726911748712167,
773
+ "loss": 0.5099314212799072,
774
+ "mean_token_accuracy": 0.848412600159645,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5583519090053647,
781
+ "eval_loss": 0.530483603477478,
782
+ "eval_mean_token_accuracy": 0.8500003374593202,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 86.3961,
785
+ "eval_samples_per_second": 15.927,
786
+ "eval_steps_per_second": 1.991,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.5454779766499996,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.890394926071167,
793
+ "learning_rate": 0.00019674973795318548,
794
+ "loss": 0.4931994915008545,
795
+ "mean_token_accuracy": 0.8540832489728928,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.572755502406941,
802
+ "eval_loss": 0.5415747761726379,
803
+ "eval_mean_token_accuracy": 0.8444425803284312,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4323,
806
+ "eval_samples_per_second": 15.92,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.5392089951783419,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.632411777973175,
814
+ "learning_rate": 0.00019620141421930058,
815
+ "loss": 0.4957888603210449,
816
+ "mean_token_accuracy": 0.8549866065382957,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.540764772961306,
823
+ "eval_loss": 0.5327216386795044,
824
+ "eval_mean_token_accuracy": 0.850631088364956,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.8097,
827
+ "eval_samples_per_second": 15.851,
828
+ "eval_steps_per_second": 1.981,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5674678739160299,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6958843469619751,
835
+ "learning_rate": 0.0001956243116524263,
836
+ "loss": 0.504389762878418,
837
+ "mean_token_accuracy": 0.8527948908507824,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.530262403190136,
844
+ "eval_loss": 0.5308871865272522,
845
+ "eval_mean_token_accuracy": 0.8522498046242913,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.7942,
848
+ "eval_samples_per_second": 15.854,
849
+ "eval_steps_per_second": 1.982,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4742849511213792,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.6941492557525635,
856
+ "learning_rate": 0.00019501860429882556,
857
+ "loss": 0.418599271774292,
858
+ "mean_token_accuracy": 0.8748210859604371,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.504602165069691,
865
+ "eval_loss": 0.542878270149231,
866
+ "eval_mean_token_accuracy": 0.8507604484641275,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.7841,
869
+ "eval_samples_per_second": 15.855,
870
+ "eval_steps_per_second": 1.982,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.45857742577791216,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.5791997909545898,
877
+ "learning_rate": 0.00019438447483157478,
878
+ "loss": 0.399777889251709,
879
+ "mean_token_accuracy": 0.8754058346152306,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.5028848362176918,
886
+ "eval_loss": 0.5356478095054626,
887
+ "eval_mean_token_accuracy": 0.8525635412959165,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.6707,
890
+ "eval_samples_per_second": 15.876,
891
+ "eval_steps_per_second": 1.985,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4869446292519569,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6483516693115234,
898
+ "learning_rate": 0.00019372211449547223,
899
+ "loss": 0.40715818405151366,
900
+ "mean_token_accuracy": 0.875113020837307,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.4928991326759028,
907
+ "eval_loss": 0.5419561862945557,
908
+ "eval_mean_token_accuracy": 0.8516040146350861,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 87.0686,
911
+ "eval_samples_per_second": 15.804,
912
+ "eval_steps_per_second": 1.975,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.45819590501487256,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.6661920547485352,
919
+ "learning_rate": 0.00019303172304936108,
920
+ "loss": 0.39511430263519287,
921
+ "mean_token_accuracy": 0.8780680045485496,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.48602560647698334,
928
+ "eval_loss": 0.5436084866523743,
929
+ "eval_mean_token_accuracy": 0.8500938470973525,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.6809,
932
+ "eval_samples_per_second": 15.874,
933
+ "eval_steps_per_second": 1.984,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4780638810247183,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6870484352111816,
940
+ "learning_rate": 0.0001923135087058851,
941
+ "loss": 0.4061615467071533,
942
+ "mean_token_accuracy": 0.8766494184732437,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.48236206035281337,
949
+ "eval_loss": 0.5446090698242188,
950
+ "eval_mean_token_accuracy": 0.8507725513258646,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.7398,
953
+ "eval_samples_per_second": 15.864,
954
+ "eval_steps_per_second": 1.983,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.463029869645834,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.6894590854644775,
961
+ "learning_rate": 0.00019156768806869427,
962
+ "loss": 0.39602413177490237,
963
+ "mean_token_accuracy": 0.876420046389103,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.4904779093556626,
970
+ "eval_loss": 0.5404934287071228,
971
+ "eval_mean_token_accuracy": 0.852238280828609,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.5348,
974
+ "eval_samples_per_second": 15.901,
975
+ "eval_steps_per_second": 1.988,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4817025110125542,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7756227254867554,
982
+ "learning_rate": 0.00019079448606712033,
983
+ "loss": 0.4177968502044678,
984
+ "mean_token_accuracy": 0.8712256088852882,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5153802815218305,
991
+ "eval_loss": 0.5424937605857849,
992
+ "eval_mean_token_accuracy": 0.8506565759348315,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.8973,
995
+ "eval_samples_per_second": 15.835,
996
+ "eval_steps_per_second": 1.979,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.46456389091908934,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 1.2000319957733154,
1003
+ "learning_rate": 0.00018999413588834105,
1004
+ "loss": 0.4084665775299072,
1005
+ "mean_token_accuracy": 0.8750658087432385,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4849439303195754,
1012
+ "eval_loss": 0.545662522315979,
1013
+ "eval_mean_token_accuracy": 0.8491013112456299,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.9049,
1016
+ "eval_samples_per_second": 15.833,
1017
+ "eval_steps_per_second": 1.979,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.4857471022754908,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.9696341753005981,
1024
+ "learning_rate": 0.0001891668789070541,
1025
+ "loss": 0.4149796962738037,
1026
+ "mean_token_accuracy": 0.8704176343977451,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.4872790058684904,
1033
+ "eval_loss": 0.5412707924842834,
1034
+ "eval_mean_token_accuracy": 0.8509329602468846,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.7846,
1037
+ "eval_samples_per_second": 15.855,
1038
+ "eval_steps_per_second": 1.982,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.4727417893707752,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.7852500677108765,
1045
+ "learning_rate": 0.0001883129646126818,
1046
+ "loss": 0.4142886161804199,
1047
+ "mean_token_accuracy": 0.8712429471313954,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5386548059624295,
1054
+ "eval_loss": 0.536101222038269,
1055
+ "eval_mean_token_accuracy": 0.8499491239009902,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.9501,
1058
+ "eval_samples_per_second": 15.825,
1059
+ "eval_steps_per_second": 1.978,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.4673406321555376,
1064
+ "epoch": 2.5379825653798256,
1065
+ "grad_norm": 0.7133921384811401,
1066
+ "learning_rate": 0.0001874326505341286,
1067
+ "loss": 0.40857529640197754,
1068
+ "mean_token_accuracy": 0.8747925907373428,
1069
+ "num_tokens": 2384270.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.5379825653798256,
1074
+ "eval_entropy": 0.495788364909416,
1075
+ "eval_loss": 0.5418923497200012,
1076
+ "eval_mean_token_accuracy": 0.851321972040243,
1077
+ "eval_num_tokens": 2384270.0,
1078
+ "eval_runtime": 86.7154,
1079
+ "eval_samples_per_second": 15.868,
1080
+ "eval_steps_per_second": 1.983,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.47599745728075504,
1085
+ "epoch": 2.587795765877958,
1086
+ "grad_norm": 0.8202953338623047,
1087
+ "learning_rate": 0.0001865262021621137,
1088
+ "loss": 0.40998234748840334,
1089
+ "mean_token_accuracy": 0.8758242674171924,
1090
+ "num_tokens": 2428036.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.587795765877958,
1095
+ "eval_entropy": 0.4887966953737791,
1096
+ "eval_loss": 0.5408804416656494,
1097
+ "eval_mean_token_accuracy": 0.8512661065473113,
1098
+ "eval_num_tokens": 2428036.0,
1099
+ "eval_runtime": 86.7869,
1100
+ "eval_samples_per_second": 15.855,
1101
+ "eval_steps_per_second": 1.982,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.4824396539479494,
1106
+ "epoch": 2.6376089663760895,
1107
+ "grad_norm": 0.6507360935211182,
1108
+ "learning_rate": 0.00018559389286910275,
1109
+ "loss": 0.4165764808654785,
1110
+ "mean_token_accuracy": 0.8722914069890976,
1111
+ "num_tokens": 2476815.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.6376089663760895,
1116
+ "eval_entropy": 0.4793398808254752,
1117
+ "eval_loss": 0.5326959490776062,
1118
+ "eval_mean_token_accuracy": 0.8534493650807891,
1119
+ "eval_num_tokens": 2476815.0,
1120
+ "eval_runtime": 86.9559,
1121
+ "eval_samples_per_second": 15.824,
1122
+ "eval_steps_per_second": 1.978,
1123
+ "step": 1060
1124
+ }
1125
+ ],
1126
+ "logging_steps": 20,
1127
+ "max_steps": 4020,
1128
+ "num_input_tokens_seen": 0,
1129
+ "num_train_epochs": 10,
1130
+ "save_steps": 20,
1131
+ "stateful_callbacks": {
1132
+ "TrainerControl": {
1133
+ "args": {
1134
+ "should_epoch_stop": false,
1135
+ "should_evaluate": false,
1136
+ "should_log": false,
1137
+ "should_save": true,
1138
+ "should_training_stop": false
1139
+ },
1140
+ "attributes": {}
1141
+ }
1142
+ },
1143
+ "total_flos": 1.045743734380032e+17,
1144
+ "train_batch_size": 4,
1145
+ "trial_name": null,
1146
+ "trial_params": null
1147
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1080/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1080/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0005183818805460705,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1080/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1080/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1080/trainer_state.json ADDED
@@ -0,0 +1,1168 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.6874221668742218,
6
+ "eval_steps": 20,
7
+ "global_step": 1080,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.9784346982836722,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.0229668617248535,
16
+ "learning_rate": 9.526142962415369e-06,
17
+ "loss": 1.7360023498535155,
18
+ "mean_token_accuracy": 0.6449888605624438,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.41506897571475,
25
+ "eval_loss": 1.1876318454742432,
26
+ "eval_mean_token_accuracy": 0.734131895525511,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.8071,
29
+ "eval_samples_per_second": 15.671,
30
+ "eval_steps_per_second": 1.959,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.049924298375845,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.5795097351074219,
37
+ "learning_rate": 1.9553661870221022e-05,
38
+ "loss": 0.8944448471069336,
39
+ "mean_token_accuracy": 0.7748479396104813,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7996658658565476,
46
+ "eval_loss": 0.7202735543251038,
47
+ "eval_mean_token_accuracy": 0.8070558306089667,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.9199,
50
+ "eval_samples_per_second": 15.831,
51
+ "eval_steps_per_second": 1.979,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7734908878803253,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3136248588562012,
58
+ "learning_rate": 2.9581180778026673e-05,
59
+ "loss": 0.6780608654022217,
60
+ "mean_token_accuracy": 0.8168170280754566,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7119324009778888,
67
+ "eval_loss": 0.6554311513900757,
68
+ "eval_mean_token_accuracy": 0.8215604798738346,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.8692,
71
+ "eval_samples_per_second": 15.84,
72
+ "eval_steps_per_second": 1.98,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7071127541363239,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.387060284614563,
79
+ "learning_rate": 3.960869968583232e-05,
80
+ "loss": 0.6382100582122803,
81
+ "mean_token_accuracy": 0.8229366384446621,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6883931482254073,
88
+ "eval_loss": 0.625065803527832,
89
+ "eval_mean_token_accuracy": 0.828940509710201,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.662,
92
+ "eval_samples_per_second": 15.878,
93
+ "eval_steps_per_second": 1.985,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6800824083387852,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9892916679382324,
100
+ "learning_rate": 4.963621859363797e-05,
101
+ "loss": 0.6011715888977051,
102
+ "mean_token_accuracy": 0.8323964163661003,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6840810470802839,
109
+ "eval_loss": 0.6037028431892395,
110
+ "eval_mean_token_accuracy": 0.8309669033732525,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.4637,
113
+ "eval_samples_per_second": 15.914,
114
+ "eval_steps_per_second": 1.989,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6776216626167297,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.8918434977531433,
121
+ "learning_rate": 5.9663737501443624e-05,
122
+ "loss": 0.5991742610931396,
123
+ "mean_token_accuracy": 0.8300838828086853,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.690427724705186,
130
+ "eval_loss": 0.5939701795578003,
131
+ "eval_mean_token_accuracy": 0.8345950186945671,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.6626,
134
+ "eval_samples_per_second": 15.878,
135
+ "eval_steps_per_second": 1.985,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6709842771291733,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9135531187057495,
142
+ "learning_rate": 6.969125640924927e-05,
143
+ "loss": 0.5914147377014161,
144
+ "mean_token_accuracy": 0.8314545609056949,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6584504666023476,
151
+ "eval_loss": 0.5849721431732178,
152
+ "eval_mean_token_accuracy": 0.8357757236375365,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.3262,
155
+ "eval_samples_per_second": 15.94,
156
+ "eval_steps_per_second": 1.992,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.6524647936224938,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8651587963104248,
163
+ "learning_rate": 7.971877531705493e-05,
164
+ "loss": 0.5710843563079834,
165
+ "mean_token_accuracy": 0.8396127380430698,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6283470298661742,
172
+ "eval_loss": 0.5738973617553711,
173
+ "eval_mean_token_accuracy": 0.8379981181649274,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.5619,
176
+ "eval_samples_per_second": 15.896,
177
+ "eval_steps_per_second": 1.987,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6450445972383022,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8661723732948303,
184
+ "learning_rate": 8.974629422486058e-05,
185
+ "loss": 0.5677794933319091,
186
+ "mean_token_accuracy": 0.8389350369572639,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6142613257086554,
193
+ "eval_loss": 0.5698265433311462,
194
+ "eval_mean_token_accuracy": 0.8388577273418737,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.4443,
197
+ "eval_samples_per_second": 15.918,
198
+ "eval_steps_per_second": 1.99,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6448334597051144,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9662242531776428,
205
+ "learning_rate": 9.977381313266624e-05,
206
+ "loss": 0.581433916091919,
207
+ "mean_token_accuracy": 0.8387043006718159,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6154296522916749,
214
+ "eval_loss": 0.5660303831100464,
215
+ "eval_mean_token_accuracy": 0.8412494766850804,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3063,
218
+ "eval_samples_per_second": 15.943,
219
+ "eval_steps_per_second": 1.993,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6376728117465973,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7618638873100281,
226
+ "learning_rate": 0.00010980133204047189,
227
+ "loss": 0.5678351402282715,
228
+ "mean_token_accuracy": 0.8404546812176704,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6181817033956217,
235
+ "eval_loss": 0.5663750171661377,
236
+ "eval_mean_token_accuracy": 0.8388350962899452,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.5904,
239
+ "eval_samples_per_second": 15.891,
240
+ "eval_steps_per_second": 1.986,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6303176879882812,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.7571695446968079,
247
+ "learning_rate": 0.00011982885094827753,
248
+ "loss": 0.5502778053283691,
249
+ "mean_token_accuracy": 0.8429657347500324,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6252533817707107,
256
+ "eval_loss": 0.5570284128189087,
257
+ "eval_mean_token_accuracy": 0.8427327847064927,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.4157,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.99,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6202544964849949,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.6447190642356873,
268
+ "learning_rate": 0.00012985636985608318,
269
+ "loss": 0.5485352993011474,
270
+ "mean_token_accuracy": 0.844165726006031,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.6441633552312851,
277
+ "eval_loss": 0.5606644153594971,
278
+ "eval_mean_token_accuracy": 0.842403513054515,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.6343,
281
+ "eval_samples_per_second": 15.883,
282
+ "eval_steps_per_second": 1.985,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6306711677461863,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7869907021522522,
289
+ "learning_rate": 0.00013988388876388883,
290
+ "loss": 0.5579307556152344,
291
+ "mean_token_accuracy": 0.841247134655714,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.6263934678809587,
298
+ "eval_loss": 0.5559113025665283,
299
+ "eval_mean_token_accuracy": 0.8427334743183713,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.6403,
302
+ "eval_samples_per_second": 15.882,
303
+ "eval_steps_per_second": 1.985,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6385872110724449,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6679229736328125,
310
+ "learning_rate": 0.0001499114076716945,
311
+ "loss": 0.5667279720306396,
312
+ "mean_token_accuracy": 0.8389136254787445,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6141417321077612,
319
+ "eval_loss": 0.5570600628852844,
320
+ "eval_mean_token_accuracy": 0.8437647996253745,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.7588,
323
+ "eval_samples_per_second": 15.86,
324
+ "eval_steps_per_second": 1.983,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6199494235217571,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.7924400568008423,
331
+ "learning_rate": 0.00015993892657950015,
332
+ "loss": 0.5529299736022949,
333
+ "mean_token_accuracy": 0.8426973208785057,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6133768925833147,
340
+ "eval_loss": 0.556602418422699,
341
+ "eval_mean_token_accuracy": 0.8432947965555413,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.492,
344
+ "eval_samples_per_second": 15.909,
345
+ "eval_steps_per_second": 1.989,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6203986253589392,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8364354372024536,
352
+ "learning_rate": 0.00016996644548730578,
353
+ "loss": 0.5551144123077393,
354
+ "mean_token_accuracy": 0.8432973213493824,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6017442844634833,
361
+ "eval_loss": 0.5566568374633789,
362
+ "eval_mean_token_accuracy": 0.8437666123689607,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.5552,
365
+ "eval_samples_per_second": 15.897,
366
+ "eval_steps_per_second": 1.987,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6341533534228802,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.7783445715904236,
373
+ "learning_rate": 0.00017999396439511144,
374
+ "loss": 0.5669133186340332,
375
+ "mean_token_accuracy": 0.8379446342587471,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6055107958788095,
382
+ "eval_loss": 0.5599350333213806,
383
+ "eval_mean_token_accuracy": 0.8435030894917112,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.4814,
386
+ "eval_samples_per_second": 15.911,
387
+ "eval_steps_per_second": 1.989,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.6306198488920927,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.8449786901473999,
394
+ "learning_rate": 0.0001900214833029171,
395
+ "loss": 0.5739435195922852,
396
+ "mean_token_accuracy": 0.8393832489848136,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6129532439071078,
403
+ "eval_loss": 0.5566295981407166,
404
+ "eval_mean_token_accuracy": 0.8430350880290187,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.4643,
407
+ "eval_samples_per_second": 15.914,
408
+ "eval_steps_per_second": 1.989,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6203123550862074,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 0.7334314584732056,
415
+ "learning_rate": 0.00020004900221072276,
416
+ "loss": 0.5547565937042236,
417
+ "mean_token_accuracy": 0.8403573960065842,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6275761647279873,
424
+ "eval_loss": 0.5621116757392883,
425
+ "eval_mean_token_accuracy": 0.841587379228237,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.4748,
428
+ "eval_samples_per_second": 15.912,
429
+ "eval_steps_per_second": 1.989,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5795013002860241,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 0.8858296871185303,
436
+ "learning_rate": 0.0002015421505577756,
437
+ "loss": 0.5183939933776855,
438
+ "mean_token_accuracy": 0.850081592034071,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5583065545489622,
445
+ "eval_loss": 0.5605642199516296,
446
+ "eval_mean_token_accuracy": 0.8439708411000496,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.5422,
449
+ "eval_samples_per_second": 15.9,
450
+ "eval_steps_per_second": 1.987,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5671238023787737,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.6882498264312744,
457
+ "learning_rate": 0.00020150112347025443,
458
+ "loss": 0.5077326774597168,
459
+ "mean_token_accuracy": 0.8489868573844432,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5868900277933409,
466
+ "eval_loss": 0.5602695345878601,
467
+ "eval_mean_token_accuracy": 0.8428842161977014,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.623,
470
+ "eval_samples_per_second": 15.885,
471
+ "eval_steps_per_second": 1.986,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5533561781048775,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 0.7717723250389099,
478
+ "learning_rate": 0.0002014297192297181,
479
+ "loss": 0.4954517364501953,
480
+ "mean_token_accuracy": 0.8529035650193691,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5600803743961246,
487
+ "eval_loss": 0.5608077645301819,
488
+ "eval_mean_token_accuracy": 0.8445036771685578,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.1316,
491
+ "eval_samples_per_second": 15.976,
492
+ "eval_steps_per_second": 1.997,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5692154694348573,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 0.7322827577590942,
499
+ "learning_rate": 0.0002013279593707117,
500
+ "loss": 0.505049467086792,
501
+ "mean_token_accuracy": 0.8551576808094978,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5732695829383162,
508
+ "eval_loss": 0.5594323873519897,
509
+ "eval_mean_token_accuracy": 0.8449713407560836,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.2726,
512
+ "eval_samples_per_second": 15.949,
513
+ "eval_steps_per_second": 1.994,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5817618492990733,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 1.1776764392852783,
520
+ "learning_rate": 0.0002011958745826208,
521
+ "loss": 0.5137609958648681,
522
+ "mean_token_accuracy": 0.8521522544324398,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5662581343636957,
529
+ "eval_loss": 0.5595026016235352,
530
+ "eval_mean_token_accuracy": 0.8441977164773053,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7261,
533
+ "eval_samples_per_second": 15.866,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5712925456464291,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7960361838340759,
541
+ "learning_rate": 0.0002010335047004159,
542
+ "loss": 0.5134767532348633,
543
+ "mean_token_accuracy": 0.8513577707111836,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.5441222797299541,
550
+ "eval_loss": 0.5535460114479065,
551
+ "eval_mean_token_accuracy": 0.8450886118550633,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.2675,
554
+ "eval_samples_per_second": 15.95,
555
+ "eval_steps_per_second": 1.994,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5787045754492283,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9205410480499268,
562
+ "learning_rate": 0.00020084089869263887,
563
+ "loss": 0.5119701862335205,
564
+ "mean_token_accuracy": 0.8503516331315041,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.5744457827057949,
571
+ "eval_loss": 0.5514978766441345,
572
+ "eval_mean_token_accuracy": 0.845929987901865,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.2299,
575
+ "eval_samples_per_second": 15.957,
576
+ "eval_steps_per_second": 1.995,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5739392962306737,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.7475653886795044,
583
+ "learning_rate": 0.00020061811464663464,
584
+ "loss": 0.5189042091369629,
585
+ "mean_token_accuracy": 0.8492388024926185,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6116398271433142,
592
+ "eval_loss": 0.551732063293457,
593
+ "eval_mean_token_accuracy": 0.8450756967067719,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.6081,
596
+ "eval_samples_per_second": 15.888,
597
+ "eval_steps_per_second": 1.986,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5755622573196888,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.8218411803245544,
604
+ "learning_rate": 0.00020036521975103286,
605
+ "loss": 0.5106248378753662,
606
+ "mean_token_accuracy": 0.8506785586476326,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.5906928708386976,
613
+ "eval_loss": 0.551278829574585,
614
+ "eval_mean_token_accuracy": 0.8462819308042526,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.5438,
617
+ "eval_samples_per_second": 15.899,
618
+ "eval_steps_per_second": 1.987,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5694822132587433,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.8880652189254761,
625
+ "learning_rate": 0.00020008229027548475,
626
+ "loss": 0.5140334606170655,
627
+ "mean_token_accuracy": 0.8521522797644139,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5599641964532608,
634
+ "eval_loss": 0.5501875877380371,
635
+ "eval_mean_token_accuracy": 0.8467660788879838,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6458,
638
+ "eval_samples_per_second": 15.881,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.5675108034163714,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.837087094783783,
646
+ "learning_rate": 0.0001997694115476612,
647
+ "loss": 0.5099846363067627,
648
+ "mean_token_accuracy": 0.8543680295348167,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5728072581249614,
655
+ "eval_loss": 0.5445425510406494,
656
+ "eval_mean_token_accuracy": 0.8474342175001321,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.4859,
659
+ "eval_samples_per_second": 15.91,
660
+ "eval_steps_per_second": 1.989,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5700885068625212,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6598765850067139,
667
+ "learning_rate": 0.000199426677927519,
668
+ "loss": 0.5122694969177246,
669
+ "mean_token_accuracy": 0.8519927568733692,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5476993622128353,
676
+ "eval_loss": 0.5427973866462708,
677
+ "eval_mean_token_accuracy": 0.8478512147138285,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.4172,
680
+ "eval_samples_per_second": 15.923,
681
+ "eval_steps_per_second": 1.99,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.5829229176044464,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.6965194940567017,
688
+ "learning_rate": 0.00019905419277884342,
689
+ "loss": 0.5253659725189209,
690
+ "mean_token_accuracy": 0.8493309423327446,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5666290084983028,
697
+ "eval_loss": 0.5467478036880493,
698
+ "eval_mean_token_accuracy": 0.8479407703460649,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.4414,
701
+ "eval_samples_per_second": 15.918,
702
+ "eval_steps_per_second": 1.99,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5498311135917902,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.636583685874939,
709
+ "learning_rate": 0.00019865206843807482,
710
+ "loss": 0.49981012344360354,
711
+ "mean_token_accuracy": 0.8560848504304885,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.539117265406043,
718
+ "eval_loss": 0.53994220495224,
719
+ "eval_mean_token_accuracy": 0.8488582601380903,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.5296,
722
+ "eval_samples_per_second": 15.902,
723
+ "eval_steps_per_second": 1.988,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5543891470879316,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.6068442463874817,
730
+ "learning_rate": 0.0001982204261804297,
731
+ "loss": 0.498047399520874,
732
+ "mean_token_accuracy": 0.8554679051041603,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5703774151760478,
739
+ "eval_loss": 0.5300245881080627,
740
+ "eval_mean_token_accuracy": 0.850798153946566,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.6456,
743
+ "eval_samples_per_second": 15.881,
744
+ "eval_steps_per_second": 1.985,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.546524541825056,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7274155020713806,
751
+ "learning_rate": 0.00019775939618332566,
752
+ "loss": 0.4988589286804199,
753
+ "mean_token_accuracy": 0.853422473371029,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5614905688305234,
760
+ "eval_loss": 0.5350332260131836,
761
+ "eval_mean_token_accuracy": 0.8492204359797544,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 86.7581,
764
+ "eval_samples_per_second": 15.86,
765
+ "eval_steps_per_second": 1.983,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5519792139530182,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.663466215133667,
772
+ "learning_rate": 0.00019726911748712167,
773
+ "loss": 0.5099314212799072,
774
+ "mean_token_accuracy": 0.848412600159645,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5583519090053647,
781
+ "eval_loss": 0.530483603477478,
782
+ "eval_mean_token_accuracy": 0.8500003374593202,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 86.3961,
785
+ "eval_samples_per_second": 15.927,
786
+ "eval_steps_per_second": 1.991,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.5454779766499996,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.890394926071167,
793
+ "learning_rate": 0.00019674973795318548,
794
+ "loss": 0.4931994915008545,
795
+ "mean_token_accuracy": 0.8540832489728928,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.572755502406941,
802
+ "eval_loss": 0.5415747761726379,
803
+ "eval_mean_token_accuracy": 0.8444425803284312,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4323,
806
+ "eval_samples_per_second": 15.92,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.5392089951783419,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.632411777973175,
814
+ "learning_rate": 0.00019620141421930058,
815
+ "loss": 0.4957888603210449,
816
+ "mean_token_accuracy": 0.8549866065382957,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.540764772961306,
823
+ "eval_loss": 0.5327216386795044,
824
+ "eval_mean_token_accuracy": 0.850631088364956,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.8097,
827
+ "eval_samples_per_second": 15.851,
828
+ "eval_steps_per_second": 1.981,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5674678739160299,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6958843469619751,
835
+ "learning_rate": 0.0001956243116524263,
836
+ "loss": 0.504389762878418,
837
+ "mean_token_accuracy": 0.8527948908507824,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.530262403190136,
844
+ "eval_loss": 0.5308871865272522,
845
+ "eval_mean_token_accuracy": 0.8522498046242913,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.7942,
848
+ "eval_samples_per_second": 15.854,
849
+ "eval_steps_per_second": 1.982,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4742849511213792,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.6941492557525635,
856
+ "learning_rate": 0.00019501860429882556,
857
+ "loss": 0.418599271774292,
858
+ "mean_token_accuracy": 0.8748210859604371,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.504602165069691,
865
+ "eval_loss": 0.542878270149231,
866
+ "eval_mean_token_accuracy": 0.8507604484641275,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.7841,
869
+ "eval_samples_per_second": 15.855,
870
+ "eval_steps_per_second": 1.982,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.45857742577791216,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.5791997909545898,
877
+ "learning_rate": 0.00019438447483157478,
878
+ "loss": 0.399777889251709,
879
+ "mean_token_accuracy": 0.8754058346152306,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.5028848362176918,
886
+ "eval_loss": 0.5356478095054626,
887
+ "eval_mean_token_accuracy": 0.8525635412959165,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.6707,
890
+ "eval_samples_per_second": 15.876,
891
+ "eval_steps_per_second": 1.985,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4869446292519569,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6483516693115234,
898
+ "learning_rate": 0.00019372211449547223,
899
+ "loss": 0.40715818405151366,
900
+ "mean_token_accuracy": 0.875113020837307,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.4928991326759028,
907
+ "eval_loss": 0.5419561862945557,
908
+ "eval_mean_token_accuracy": 0.8516040146350861,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 87.0686,
911
+ "eval_samples_per_second": 15.804,
912
+ "eval_steps_per_second": 1.975,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.45819590501487256,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.6661920547485352,
919
+ "learning_rate": 0.00019303172304936108,
920
+ "loss": 0.39511430263519287,
921
+ "mean_token_accuracy": 0.8780680045485496,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.48602560647698334,
928
+ "eval_loss": 0.5436084866523743,
929
+ "eval_mean_token_accuracy": 0.8500938470973525,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.6809,
932
+ "eval_samples_per_second": 15.874,
933
+ "eval_steps_per_second": 1.984,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4780638810247183,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6870484352111816,
940
+ "learning_rate": 0.0001923135087058851,
941
+ "loss": 0.4061615467071533,
942
+ "mean_token_accuracy": 0.8766494184732437,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.48236206035281337,
949
+ "eval_loss": 0.5446090698242188,
950
+ "eval_mean_token_accuracy": 0.8507725513258646,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.7398,
953
+ "eval_samples_per_second": 15.864,
954
+ "eval_steps_per_second": 1.983,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.463029869645834,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.6894590854644775,
961
+ "learning_rate": 0.00019156768806869427,
962
+ "loss": 0.39602413177490237,
963
+ "mean_token_accuracy": 0.876420046389103,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.4904779093556626,
970
+ "eval_loss": 0.5404934287071228,
971
+ "eval_mean_token_accuracy": 0.852238280828609,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.5348,
974
+ "eval_samples_per_second": 15.901,
975
+ "eval_steps_per_second": 1.988,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4817025110125542,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7756227254867554,
982
+ "learning_rate": 0.00019079448606712033,
983
+ "loss": 0.4177968502044678,
984
+ "mean_token_accuracy": 0.8712256088852882,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5153802815218305,
991
+ "eval_loss": 0.5424937605857849,
992
+ "eval_mean_token_accuracy": 0.8506565759348315,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.8973,
995
+ "eval_samples_per_second": 15.835,
996
+ "eval_steps_per_second": 1.979,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.46456389091908934,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 1.2000319957733154,
1003
+ "learning_rate": 0.00018999413588834105,
1004
+ "loss": 0.4084665775299072,
1005
+ "mean_token_accuracy": 0.8750658087432385,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4849439303195754,
1012
+ "eval_loss": 0.545662522315979,
1013
+ "eval_mean_token_accuracy": 0.8491013112456299,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.9049,
1016
+ "eval_samples_per_second": 15.833,
1017
+ "eval_steps_per_second": 1.979,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.4857471022754908,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.9696341753005981,
1024
+ "learning_rate": 0.0001891668789070541,
1025
+ "loss": 0.4149796962738037,
1026
+ "mean_token_accuracy": 0.8704176343977451,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.4872790058684904,
1033
+ "eval_loss": 0.5412707924842834,
1034
+ "eval_mean_token_accuracy": 0.8509329602468846,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.7846,
1037
+ "eval_samples_per_second": 15.855,
1038
+ "eval_steps_per_second": 1.982,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.4727417893707752,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.7852500677108765,
1045
+ "learning_rate": 0.0001883129646126818,
1046
+ "loss": 0.4142886161804199,
1047
+ "mean_token_accuracy": 0.8712429471313954,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5386548059624295,
1054
+ "eval_loss": 0.536101222038269,
1055
+ "eval_mean_token_accuracy": 0.8499491239009902,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.9501,
1058
+ "eval_samples_per_second": 15.825,
1059
+ "eval_steps_per_second": 1.978,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.4673406321555376,
1064
+ "epoch": 2.5379825653798256,
1065
+ "grad_norm": 0.7133921384811401,
1066
+ "learning_rate": 0.0001874326505341286,
1067
+ "loss": 0.40857529640197754,
1068
+ "mean_token_accuracy": 0.8747925907373428,
1069
+ "num_tokens": 2384270.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.5379825653798256,
1074
+ "eval_entropy": 0.495788364909416,
1075
+ "eval_loss": 0.5418923497200012,
1076
+ "eval_mean_token_accuracy": 0.851321972040243,
1077
+ "eval_num_tokens": 2384270.0,
1078
+ "eval_runtime": 86.7154,
1079
+ "eval_samples_per_second": 15.868,
1080
+ "eval_steps_per_second": 1.983,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.47599745728075504,
1085
+ "epoch": 2.587795765877958,
1086
+ "grad_norm": 0.8202953338623047,
1087
+ "learning_rate": 0.0001865262021621137,
1088
+ "loss": 0.40998234748840334,
1089
+ "mean_token_accuracy": 0.8758242674171924,
1090
+ "num_tokens": 2428036.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.587795765877958,
1095
+ "eval_entropy": 0.4887966953737791,
1096
+ "eval_loss": 0.5408804416656494,
1097
+ "eval_mean_token_accuracy": 0.8512661065473113,
1098
+ "eval_num_tokens": 2428036.0,
1099
+ "eval_runtime": 86.7869,
1100
+ "eval_samples_per_second": 15.855,
1101
+ "eval_steps_per_second": 1.982,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.4824396539479494,
1106
+ "epoch": 2.6376089663760895,
1107
+ "grad_norm": 0.6507360935211182,
1108
+ "learning_rate": 0.00018559389286910275,
1109
+ "loss": 0.4165764808654785,
1110
+ "mean_token_accuracy": 0.8722914069890976,
1111
+ "num_tokens": 2476815.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.6376089663760895,
1116
+ "eval_entropy": 0.4793398808254752,
1117
+ "eval_loss": 0.5326959490776062,
1118
+ "eval_mean_token_accuracy": 0.8534493650807891,
1119
+ "eval_num_tokens": 2476815.0,
1120
+ "eval_runtime": 86.9559,
1121
+ "eval_samples_per_second": 15.824,
1122
+ "eval_steps_per_second": 1.978,
1123
+ "step": 1060
1124
+ },
1125
+ {
1126
+ "entropy": 0.4605010639876127,
1127
+ "epoch": 2.6874221668742218,
1128
+ "grad_norm": 0.6740535497665405,
1129
+ "learning_rate": 0.00018463600382686253,
1130
+ "loss": 0.4123940944671631,
1131
+ "mean_token_accuracy": 0.8733638986945153,
1132
+ "num_tokens": 2527131.0,
1133
+ "step": 1080
1134
+ },
1135
+ {
1136
+ "epoch": 2.6874221668742218,
1137
+ "eval_entropy": 0.47902208583992584,
1138
+ "eval_loss": 0.5372340083122253,
1139
+ "eval_mean_token_accuracy": 0.851325950303743,
1140
+ "eval_num_tokens": 2527131.0,
1141
+ "eval_runtime": 86.9638,
1142
+ "eval_samples_per_second": 15.823,
1143
+ "eval_steps_per_second": 1.978,
1144
+ "step": 1080
1145
+ }
1146
+ ],
1147
+ "logging_steps": 20,
1148
+ "max_steps": 4020,
1149
+ "num_input_tokens_seen": 0,
1150
+ "num_train_epochs": 10,
1151
+ "save_steps": 20,
1152
+ "stateful_callbacks": {
1153
+ "TrainerControl": {
1154
+ "args": {
1155
+ "should_epoch_stop": false,
1156
+ "should_evaluate": false,
1157
+ "should_log": false,
1158
+ "should_save": true,
1159
+ "should_training_stop": false
1160
+ },
1161
+ "attributes": {}
1162
+ }
1163
+ },
1164
+ "total_flos": 1.0682640451304448e+17,
1165
+ "train_batch_size": 4,
1166
+ "trial_name": null,
1167
+ "trial_params": null
1168
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1100/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1100/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0005183818805460705,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1100/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1100/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1100/trainer_state.json ADDED
@@ -0,0 +1,1189 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.7372353673723535,
6
+ "eval_steps": 20,
7
+ "global_step": 1100,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.9784346982836722,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.0229668617248535,
16
+ "learning_rate": 9.526142962415369e-06,
17
+ "loss": 1.7360023498535155,
18
+ "mean_token_accuracy": 0.6449888605624438,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.41506897571475,
25
+ "eval_loss": 1.1876318454742432,
26
+ "eval_mean_token_accuracy": 0.734131895525511,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.8071,
29
+ "eval_samples_per_second": 15.671,
30
+ "eval_steps_per_second": 1.959,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.049924298375845,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.5795097351074219,
37
+ "learning_rate": 1.9553661870221022e-05,
38
+ "loss": 0.8944448471069336,
39
+ "mean_token_accuracy": 0.7748479396104813,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7996658658565476,
46
+ "eval_loss": 0.7202735543251038,
47
+ "eval_mean_token_accuracy": 0.8070558306089667,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.9199,
50
+ "eval_samples_per_second": 15.831,
51
+ "eval_steps_per_second": 1.979,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7734908878803253,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3136248588562012,
58
+ "learning_rate": 2.9581180778026673e-05,
59
+ "loss": 0.6780608654022217,
60
+ "mean_token_accuracy": 0.8168170280754566,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7119324009778888,
67
+ "eval_loss": 0.6554311513900757,
68
+ "eval_mean_token_accuracy": 0.8215604798738346,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.8692,
71
+ "eval_samples_per_second": 15.84,
72
+ "eval_steps_per_second": 1.98,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7071127541363239,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.387060284614563,
79
+ "learning_rate": 3.960869968583232e-05,
80
+ "loss": 0.6382100582122803,
81
+ "mean_token_accuracy": 0.8229366384446621,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6883931482254073,
88
+ "eval_loss": 0.625065803527832,
89
+ "eval_mean_token_accuracy": 0.828940509710201,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.662,
92
+ "eval_samples_per_second": 15.878,
93
+ "eval_steps_per_second": 1.985,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6800824083387852,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9892916679382324,
100
+ "learning_rate": 4.963621859363797e-05,
101
+ "loss": 0.6011715888977051,
102
+ "mean_token_accuracy": 0.8323964163661003,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6840810470802839,
109
+ "eval_loss": 0.6037028431892395,
110
+ "eval_mean_token_accuracy": 0.8309669033732525,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.4637,
113
+ "eval_samples_per_second": 15.914,
114
+ "eval_steps_per_second": 1.989,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6776216626167297,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.8918434977531433,
121
+ "learning_rate": 5.9663737501443624e-05,
122
+ "loss": 0.5991742610931396,
123
+ "mean_token_accuracy": 0.8300838828086853,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.690427724705186,
130
+ "eval_loss": 0.5939701795578003,
131
+ "eval_mean_token_accuracy": 0.8345950186945671,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.6626,
134
+ "eval_samples_per_second": 15.878,
135
+ "eval_steps_per_second": 1.985,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6709842771291733,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9135531187057495,
142
+ "learning_rate": 6.969125640924927e-05,
143
+ "loss": 0.5914147377014161,
144
+ "mean_token_accuracy": 0.8314545609056949,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6584504666023476,
151
+ "eval_loss": 0.5849721431732178,
152
+ "eval_mean_token_accuracy": 0.8357757236375365,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.3262,
155
+ "eval_samples_per_second": 15.94,
156
+ "eval_steps_per_second": 1.992,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.6524647936224938,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8651587963104248,
163
+ "learning_rate": 7.971877531705493e-05,
164
+ "loss": 0.5710843563079834,
165
+ "mean_token_accuracy": 0.8396127380430698,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6283470298661742,
172
+ "eval_loss": 0.5738973617553711,
173
+ "eval_mean_token_accuracy": 0.8379981181649274,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.5619,
176
+ "eval_samples_per_second": 15.896,
177
+ "eval_steps_per_second": 1.987,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6450445972383022,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8661723732948303,
184
+ "learning_rate": 8.974629422486058e-05,
185
+ "loss": 0.5677794933319091,
186
+ "mean_token_accuracy": 0.8389350369572639,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6142613257086554,
193
+ "eval_loss": 0.5698265433311462,
194
+ "eval_mean_token_accuracy": 0.8388577273418737,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.4443,
197
+ "eval_samples_per_second": 15.918,
198
+ "eval_steps_per_second": 1.99,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6448334597051144,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9662242531776428,
205
+ "learning_rate": 9.977381313266624e-05,
206
+ "loss": 0.581433916091919,
207
+ "mean_token_accuracy": 0.8387043006718159,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6154296522916749,
214
+ "eval_loss": 0.5660303831100464,
215
+ "eval_mean_token_accuracy": 0.8412494766850804,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3063,
218
+ "eval_samples_per_second": 15.943,
219
+ "eval_steps_per_second": 1.993,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6376728117465973,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7618638873100281,
226
+ "learning_rate": 0.00010980133204047189,
227
+ "loss": 0.5678351402282715,
228
+ "mean_token_accuracy": 0.8404546812176704,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6181817033956217,
235
+ "eval_loss": 0.5663750171661377,
236
+ "eval_mean_token_accuracy": 0.8388350962899452,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.5904,
239
+ "eval_samples_per_second": 15.891,
240
+ "eval_steps_per_second": 1.986,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6303176879882812,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.7571695446968079,
247
+ "learning_rate": 0.00011982885094827753,
248
+ "loss": 0.5502778053283691,
249
+ "mean_token_accuracy": 0.8429657347500324,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6252533817707107,
256
+ "eval_loss": 0.5570284128189087,
257
+ "eval_mean_token_accuracy": 0.8427327847064927,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.4157,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.99,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6202544964849949,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.6447190642356873,
268
+ "learning_rate": 0.00012985636985608318,
269
+ "loss": 0.5485352993011474,
270
+ "mean_token_accuracy": 0.844165726006031,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.6441633552312851,
277
+ "eval_loss": 0.5606644153594971,
278
+ "eval_mean_token_accuracy": 0.842403513054515,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.6343,
281
+ "eval_samples_per_second": 15.883,
282
+ "eval_steps_per_second": 1.985,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6306711677461863,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7869907021522522,
289
+ "learning_rate": 0.00013988388876388883,
290
+ "loss": 0.5579307556152344,
291
+ "mean_token_accuracy": 0.841247134655714,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.6263934678809587,
298
+ "eval_loss": 0.5559113025665283,
299
+ "eval_mean_token_accuracy": 0.8427334743183713,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.6403,
302
+ "eval_samples_per_second": 15.882,
303
+ "eval_steps_per_second": 1.985,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6385872110724449,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6679229736328125,
310
+ "learning_rate": 0.0001499114076716945,
311
+ "loss": 0.5667279720306396,
312
+ "mean_token_accuracy": 0.8389136254787445,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6141417321077612,
319
+ "eval_loss": 0.5570600628852844,
320
+ "eval_mean_token_accuracy": 0.8437647996253745,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.7588,
323
+ "eval_samples_per_second": 15.86,
324
+ "eval_steps_per_second": 1.983,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6199494235217571,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.7924400568008423,
331
+ "learning_rate": 0.00015993892657950015,
332
+ "loss": 0.5529299736022949,
333
+ "mean_token_accuracy": 0.8426973208785057,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6133768925833147,
340
+ "eval_loss": 0.556602418422699,
341
+ "eval_mean_token_accuracy": 0.8432947965555413,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.492,
344
+ "eval_samples_per_second": 15.909,
345
+ "eval_steps_per_second": 1.989,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6203986253589392,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8364354372024536,
352
+ "learning_rate": 0.00016996644548730578,
353
+ "loss": 0.5551144123077393,
354
+ "mean_token_accuracy": 0.8432973213493824,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6017442844634833,
361
+ "eval_loss": 0.5566568374633789,
362
+ "eval_mean_token_accuracy": 0.8437666123689607,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.5552,
365
+ "eval_samples_per_second": 15.897,
366
+ "eval_steps_per_second": 1.987,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6341533534228802,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.7783445715904236,
373
+ "learning_rate": 0.00017999396439511144,
374
+ "loss": 0.5669133186340332,
375
+ "mean_token_accuracy": 0.8379446342587471,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6055107958788095,
382
+ "eval_loss": 0.5599350333213806,
383
+ "eval_mean_token_accuracy": 0.8435030894917112,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.4814,
386
+ "eval_samples_per_second": 15.911,
387
+ "eval_steps_per_second": 1.989,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.6306198488920927,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.8449786901473999,
394
+ "learning_rate": 0.0001900214833029171,
395
+ "loss": 0.5739435195922852,
396
+ "mean_token_accuracy": 0.8393832489848136,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6129532439071078,
403
+ "eval_loss": 0.5566295981407166,
404
+ "eval_mean_token_accuracy": 0.8430350880290187,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.4643,
407
+ "eval_samples_per_second": 15.914,
408
+ "eval_steps_per_second": 1.989,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6203123550862074,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 0.7334314584732056,
415
+ "learning_rate": 0.00020004900221072276,
416
+ "loss": 0.5547565937042236,
417
+ "mean_token_accuracy": 0.8403573960065842,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6275761647279873,
424
+ "eval_loss": 0.5621116757392883,
425
+ "eval_mean_token_accuracy": 0.841587379228237,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.4748,
428
+ "eval_samples_per_second": 15.912,
429
+ "eval_steps_per_second": 1.989,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5795013002860241,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 0.8858296871185303,
436
+ "learning_rate": 0.0002015421505577756,
437
+ "loss": 0.5183939933776855,
438
+ "mean_token_accuracy": 0.850081592034071,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5583065545489622,
445
+ "eval_loss": 0.5605642199516296,
446
+ "eval_mean_token_accuracy": 0.8439708411000496,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.5422,
449
+ "eval_samples_per_second": 15.9,
450
+ "eval_steps_per_second": 1.987,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5671238023787737,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.6882498264312744,
457
+ "learning_rate": 0.00020150112347025443,
458
+ "loss": 0.5077326774597168,
459
+ "mean_token_accuracy": 0.8489868573844432,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5868900277933409,
466
+ "eval_loss": 0.5602695345878601,
467
+ "eval_mean_token_accuracy": 0.8428842161977014,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.623,
470
+ "eval_samples_per_second": 15.885,
471
+ "eval_steps_per_second": 1.986,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5533561781048775,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 0.7717723250389099,
478
+ "learning_rate": 0.0002014297192297181,
479
+ "loss": 0.4954517364501953,
480
+ "mean_token_accuracy": 0.8529035650193691,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5600803743961246,
487
+ "eval_loss": 0.5608077645301819,
488
+ "eval_mean_token_accuracy": 0.8445036771685578,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.1316,
491
+ "eval_samples_per_second": 15.976,
492
+ "eval_steps_per_second": 1.997,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5692154694348573,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 0.7322827577590942,
499
+ "learning_rate": 0.0002013279593707117,
500
+ "loss": 0.505049467086792,
501
+ "mean_token_accuracy": 0.8551576808094978,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5732695829383162,
508
+ "eval_loss": 0.5594323873519897,
509
+ "eval_mean_token_accuracy": 0.8449713407560836,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.2726,
512
+ "eval_samples_per_second": 15.949,
513
+ "eval_steps_per_second": 1.994,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5817618492990733,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 1.1776764392852783,
520
+ "learning_rate": 0.0002011958745826208,
521
+ "loss": 0.5137609958648681,
522
+ "mean_token_accuracy": 0.8521522544324398,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5662581343636957,
529
+ "eval_loss": 0.5595026016235352,
530
+ "eval_mean_token_accuracy": 0.8441977164773053,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7261,
533
+ "eval_samples_per_second": 15.866,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5712925456464291,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7960361838340759,
541
+ "learning_rate": 0.0002010335047004159,
542
+ "loss": 0.5134767532348633,
543
+ "mean_token_accuracy": 0.8513577707111836,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.5441222797299541,
550
+ "eval_loss": 0.5535460114479065,
551
+ "eval_mean_token_accuracy": 0.8450886118550633,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.2675,
554
+ "eval_samples_per_second": 15.95,
555
+ "eval_steps_per_second": 1.994,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5787045754492283,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9205410480499268,
562
+ "learning_rate": 0.00020084089869263887,
563
+ "loss": 0.5119701862335205,
564
+ "mean_token_accuracy": 0.8503516331315041,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.5744457827057949,
571
+ "eval_loss": 0.5514978766441345,
572
+ "eval_mean_token_accuracy": 0.845929987901865,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.2299,
575
+ "eval_samples_per_second": 15.957,
576
+ "eval_steps_per_second": 1.995,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5739392962306737,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.7475653886795044,
583
+ "learning_rate": 0.00020061811464663464,
584
+ "loss": 0.5189042091369629,
585
+ "mean_token_accuracy": 0.8492388024926185,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6116398271433142,
592
+ "eval_loss": 0.551732063293457,
593
+ "eval_mean_token_accuracy": 0.8450756967067719,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.6081,
596
+ "eval_samples_per_second": 15.888,
597
+ "eval_steps_per_second": 1.986,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5755622573196888,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.8218411803245544,
604
+ "learning_rate": 0.00020036521975103286,
605
+ "loss": 0.5106248378753662,
606
+ "mean_token_accuracy": 0.8506785586476326,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.5906928708386976,
613
+ "eval_loss": 0.551278829574585,
614
+ "eval_mean_token_accuracy": 0.8462819308042526,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.5438,
617
+ "eval_samples_per_second": 15.899,
618
+ "eval_steps_per_second": 1.987,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5694822132587433,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.8880652189254761,
625
+ "learning_rate": 0.00020008229027548475,
626
+ "loss": 0.5140334606170655,
627
+ "mean_token_accuracy": 0.8521522797644139,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5599641964532608,
634
+ "eval_loss": 0.5501875877380371,
635
+ "eval_mean_token_accuracy": 0.8467660788879838,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6458,
638
+ "eval_samples_per_second": 15.881,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.5675108034163714,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.837087094783783,
646
+ "learning_rate": 0.0001997694115476612,
647
+ "loss": 0.5099846363067627,
648
+ "mean_token_accuracy": 0.8543680295348167,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5728072581249614,
655
+ "eval_loss": 0.5445425510406494,
656
+ "eval_mean_token_accuracy": 0.8474342175001321,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.4859,
659
+ "eval_samples_per_second": 15.91,
660
+ "eval_steps_per_second": 1.989,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5700885068625212,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6598765850067139,
667
+ "learning_rate": 0.000199426677927519,
668
+ "loss": 0.5122694969177246,
669
+ "mean_token_accuracy": 0.8519927568733692,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5476993622128353,
676
+ "eval_loss": 0.5427973866462708,
677
+ "eval_mean_token_accuracy": 0.8478512147138285,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.4172,
680
+ "eval_samples_per_second": 15.923,
681
+ "eval_steps_per_second": 1.99,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.5829229176044464,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.6965194940567017,
688
+ "learning_rate": 0.00019905419277884342,
689
+ "loss": 0.5253659725189209,
690
+ "mean_token_accuracy": 0.8493309423327446,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5666290084983028,
697
+ "eval_loss": 0.5467478036880493,
698
+ "eval_mean_token_accuracy": 0.8479407703460649,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.4414,
701
+ "eval_samples_per_second": 15.918,
702
+ "eval_steps_per_second": 1.99,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5498311135917902,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.636583685874939,
709
+ "learning_rate": 0.00019865206843807482,
710
+ "loss": 0.49981012344360354,
711
+ "mean_token_accuracy": 0.8560848504304885,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.539117265406043,
718
+ "eval_loss": 0.53994220495224,
719
+ "eval_mean_token_accuracy": 0.8488582601380903,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.5296,
722
+ "eval_samples_per_second": 15.902,
723
+ "eval_steps_per_second": 1.988,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5543891470879316,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.6068442463874817,
730
+ "learning_rate": 0.0001982204261804297,
731
+ "loss": 0.498047399520874,
732
+ "mean_token_accuracy": 0.8554679051041603,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5703774151760478,
739
+ "eval_loss": 0.5300245881080627,
740
+ "eval_mean_token_accuracy": 0.850798153946566,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.6456,
743
+ "eval_samples_per_second": 15.881,
744
+ "eval_steps_per_second": 1.985,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.546524541825056,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7274155020713806,
751
+ "learning_rate": 0.00019775939618332566,
752
+ "loss": 0.4988589286804199,
753
+ "mean_token_accuracy": 0.853422473371029,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5614905688305234,
760
+ "eval_loss": 0.5350332260131836,
761
+ "eval_mean_token_accuracy": 0.8492204359797544,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 86.7581,
764
+ "eval_samples_per_second": 15.86,
765
+ "eval_steps_per_second": 1.983,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5519792139530182,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.663466215133667,
772
+ "learning_rate": 0.00019726911748712167,
773
+ "loss": 0.5099314212799072,
774
+ "mean_token_accuracy": 0.848412600159645,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5583519090053647,
781
+ "eval_loss": 0.530483603477478,
782
+ "eval_mean_token_accuracy": 0.8500003374593202,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 86.3961,
785
+ "eval_samples_per_second": 15.927,
786
+ "eval_steps_per_second": 1.991,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.5454779766499996,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.890394926071167,
793
+ "learning_rate": 0.00019674973795318548,
794
+ "loss": 0.4931994915008545,
795
+ "mean_token_accuracy": 0.8540832489728928,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.572755502406941,
802
+ "eval_loss": 0.5415747761726379,
803
+ "eval_mean_token_accuracy": 0.8444425803284312,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4323,
806
+ "eval_samples_per_second": 15.92,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.5392089951783419,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.632411777973175,
814
+ "learning_rate": 0.00019620141421930058,
815
+ "loss": 0.4957888603210449,
816
+ "mean_token_accuracy": 0.8549866065382957,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.540764772961306,
823
+ "eval_loss": 0.5327216386795044,
824
+ "eval_mean_token_accuracy": 0.850631088364956,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.8097,
827
+ "eval_samples_per_second": 15.851,
828
+ "eval_steps_per_second": 1.981,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5674678739160299,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6958843469619751,
835
+ "learning_rate": 0.0001956243116524263,
836
+ "loss": 0.504389762878418,
837
+ "mean_token_accuracy": 0.8527948908507824,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.530262403190136,
844
+ "eval_loss": 0.5308871865272522,
845
+ "eval_mean_token_accuracy": 0.8522498046242913,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.7942,
848
+ "eval_samples_per_second": 15.854,
849
+ "eval_steps_per_second": 1.982,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4742849511213792,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.6941492557525635,
856
+ "learning_rate": 0.00019501860429882556,
857
+ "loss": 0.418599271774292,
858
+ "mean_token_accuracy": 0.8748210859604371,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.504602165069691,
865
+ "eval_loss": 0.542878270149231,
866
+ "eval_mean_token_accuracy": 0.8507604484641275,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.7841,
869
+ "eval_samples_per_second": 15.855,
870
+ "eval_steps_per_second": 1.982,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.45857742577791216,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.5791997909545898,
877
+ "learning_rate": 0.00019438447483157478,
878
+ "loss": 0.399777889251709,
879
+ "mean_token_accuracy": 0.8754058346152306,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.5028848362176918,
886
+ "eval_loss": 0.5356478095054626,
887
+ "eval_mean_token_accuracy": 0.8525635412959165,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.6707,
890
+ "eval_samples_per_second": 15.876,
891
+ "eval_steps_per_second": 1.985,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4869446292519569,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6483516693115234,
898
+ "learning_rate": 0.00019372211449547223,
899
+ "loss": 0.40715818405151366,
900
+ "mean_token_accuracy": 0.875113020837307,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.4928991326759028,
907
+ "eval_loss": 0.5419561862945557,
908
+ "eval_mean_token_accuracy": 0.8516040146350861,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 87.0686,
911
+ "eval_samples_per_second": 15.804,
912
+ "eval_steps_per_second": 1.975,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.45819590501487256,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.6661920547485352,
919
+ "learning_rate": 0.00019303172304936108,
920
+ "loss": 0.39511430263519287,
921
+ "mean_token_accuracy": 0.8780680045485496,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.48602560647698334,
928
+ "eval_loss": 0.5436084866523743,
929
+ "eval_mean_token_accuracy": 0.8500938470973525,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.6809,
932
+ "eval_samples_per_second": 15.874,
933
+ "eval_steps_per_second": 1.984,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4780638810247183,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6870484352111816,
940
+ "learning_rate": 0.0001923135087058851,
941
+ "loss": 0.4061615467071533,
942
+ "mean_token_accuracy": 0.8766494184732437,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.48236206035281337,
949
+ "eval_loss": 0.5446090698242188,
950
+ "eval_mean_token_accuracy": 0.8507725513258646,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.7398,
953
+ "eval_samples_per_second": 15.864,
954
+ "eval_steps_per_second": 1.983,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.463029869645834,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.6894590854644775,
961
+ "learning_rate": 0.00019156768806869427,
962
+ "loss": 0.39602413177490237,
963
+ "mean_token_accuracy": 0.876420046389103,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.4904779093556626,
970
+ "eval_loss": 0.5404934287071228,
971
+ "eval_mean_token_accuracy": 0.852238280828609,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.5348,
974
+ "eval_samples_per_second": 15.901,
975
+ "eval_steps_per_second": 1.988,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4817025110125542,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7756227254867554,
982
+ "learning_rate": 0.00019079448606712033,
983
+ "loss": 0.4177968502044678,
984
+ "mean_token_accuracy": 0.8712256088852882,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5153802815218305,
991
+ "eval_loss": 0.5424937605857849,
992
+ "eval_mean_token_accuracy": 0.8506565759348315,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.8973,
995
+ "eval_samples_per_second": 15.835,
996
+ "eval_steps_per_second": 1.979,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.46456389091908934,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 1.2000319957733154,
1003
+ "learning_rate": 0.00018999413588834105,
1004
+ "loss": 0.4084665775299072,
1005
+ "mean_token_accuracy": 0.8750658087432385,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4849439303195754,
1012
+ "eval_loss": 0.545662522315979,
1013
+ "eval_mean_token_accuracy": 0.8491013112456299,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.9049,
1016
+ "eval_samples_per_second": 15.833,
1017
+ "eval_steps_per_second": 1.979,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.4857471022754908,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.9696341753005981,
1024
+ "learning_rate": 0.0001891668789070541,
1025
+ "loss": 0.4149796962738037,
1026
+ "mean_token_accuracy": 0.8704176343977451,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.4872790058684904,
1033
+ "eval_loss": 0.5412707924842834,
1034
+ "eval_mean_token_accuracy": 0.8509329602468846,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.7846,
1037
+ "eval_samples_per_second": 15.855,
1038
+ "eval_steps_per_second": 1.982,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.4727417893707752,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.7852500677108765,
1045
+ "learning_rate": 0.0001883129646126818,
1046
+ "loss": 0.4142886161804199,
1047
+ "mean_token_accuracy": 0.8712429471313954,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5386548059624295,
1054
+ "eval_loss": 0.536101222038269,
1055
+ "eval_mean_token_accuracy": 0.8499491239009902,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.9501,
1058
+ "eval_samples_per_second": 15.825,
1059
+ "eval_steps_per_second": 1.978,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.4673406321555376,
1064
+ "epoch": 2.5379825653798256,
1065
+ "grad_norm": 0.7133921384811401,
1066
+ "learning_rate": 0.0001874326505341286,
1067
+ "loss": 0.40857529640197754,
1068
+ "mean_token_accuracy": 0.8747925907373428,
1069
+ "num_tokens": 2384270.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.5379825653798256,
1074
+ "eval_entropy": 0.495788364909416,
1075
+ "eval_loss": 0.5418923497200012,
1076
+ "eval_mean_token_accuracy": 0.851321972040243,
1077
+ "eval_num_tokens": 2384270.0,
1078
+ "eval_runtime": 86.7154,
1079
+ "eval_samples_per_second": 15.868,
1080
+ "eval_steps_per_second": 1.983,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.47599745728075504,
1085
+ "epoch": 2.587795765877958,
1086
+ "grad_norm": 0.8202953338623047,
1087
+ "learning_rate": 0.0001865262021621137,
1088
+ "loss": 0.40998234748840334,
1089
+ "mean_token_accuracy": 0.8758242674171924,
1090
+ "num_tokens": 2428036.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.587795765877958,
1095
+ "eval_entropy": 0.4887966953737791,
1096
+ "eval_loss": 0.5408804416656494,
1097
+ "eval_mean_token_accuracy": 0.8512661065473113,
1098
+ "eval_num_tokens": 2428036.0,
1099
+ "eval_runtime": 86.7869,
1100
+ "eval_samples_per_second": 15.855,
1101
+ "eval_steps_per_second": 1.982,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.4824396539479494,
1106
+ "epoch": 2.6376089663760895,
1107
+ "grad_norm": 0.6507360935211182,
1108
+ "learning_rate": 0.00018559389286910275,
1109
+ "loss": 0.4165764808654785,
1110
+ "mean_token_accuracy": 0.8722914069890976,
1111
+ "num_tokens": 2476815.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.6376089663760895,
1116
+ "eval_entropy": 0.4793398808254752,
1117
+ "eval_loss": 0.5326959490776062,
1118
+ "eval_mean_token_accuracy": 0.8534493650807891,
1119
+ "eval_num_tokens": 2476815.0,
1120
+ "eval_runtime": 86.9559,
1121
+ "eval_samples_per_second": 15.824,
1122
+ "eval_steps_per_second": 1.978,
1123
+ "step": 1060
1124
+ },
1125
+ {
1126
+ "entropy": 0.4605010639876127,
1127
+ "epoch": 2.6874221668742218,
1128
+ "grad_norm": 0.6740535497665405,
1129
+ "learning_rate": 0.00018463600382686253,
1130
+ "loss": 0.4123940944671631,
1131
+ "mean_token_accuracy": 0.8733638986945153,
1132
+ "num_tokens": 2527131.0,
1133
+ "step": 1080
1134
+ },
1135
+ {
1136
+ "epoch": 2.6874221668742218,
1137
+ "eval_entropy": 0.47902208583992584,
1138
+ "eval_loss": 0.5372340083122253,
1139
+ "eval_mean_token_accuracy": 0.851325950303743,
1140
+ "eval_num_tokens": 2527131.0,
1141
+ "eval_runtime": 86.9638,
1142
+ "eval_samples_per_second": 15.823,
1143
+ "eval_steps_per_second": 1.978,
1144
+ "step": 1080
1145
+ },
1146
+ {
1147
+ "entropy": 0.4872019402682781,
1148
+ "epoch": 2.7372353673723535,
1149
+ "grad_norm": 0.6994742155075073,
1150
+ "learning_rate": 0.0001836528239216632,
1151
+ "loss": 0.41599602699279786,
1152
+ "mean_token_accuracy": 0.872775862365961,
1153
+ "num_tokens": 2572537.0,
1154
+ "step": 1100
1155
+ },
1156
+ {
1157
+ "epoch": 2.7372353673723535,
1158
+ "eval_entropy": 0.4893243626453156,
1159
+ "eval_loss": 0.5327795743942261,
1160
+ "eval_mean_token_accuracy": 0.8537560302850812,
1161
+ "eval_num_tokens": 2572537.0,
1162
+ "eval_runtime": 86.823,
1163
+ "eval_samples_per_second": 15.848,
1164
+ "eval_steps_per_second": 1.981,
1165
+ "step": 1100
1166
+ }
1167
+ ],
1168
+ "logging_steps": 20,
1169
+ "max_steps": 4020,
1170
+ "num_input_tokens_seen": 0,
1171
+ "num_train_epochs": 10,
1172
+ "save_steps": 20,
1173
+ "stateful_callbacks": {
1174
+ "TrainerControl": {
1175
+ "args": {
1176
+ "should_epoch_stop": false,
1177
+ "should_evaluate": false,
1178
+ "should_log": false,
1179
+ "should_save": true,
1180
+ "should_training_stop": false
1181
+ },
1182
+ "attributes": {}
1183
+ }
1184
+ },
1185
+ "total_flos": 1.0877774590688256e+17,
1186
+ "train_batch_size": 4,
1187
+ "trial_name": null,
1188
+ "trial_params": null
1189
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1120/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1120/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0005183818805460705,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1120/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1120/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1120/trainer_state.json ADDED
@@ -0,0 +1,1210 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.7870485678704857,
6
+ "eval_steps": 20,
7
+ "global_step": 1120,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.9784346982836722,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.0229668617248535,
16
+ "learning_rate": 9.526142962415369e-06,
17
+ "loss": 1.7360023498535155,
18
+ "mean_token_accuracy": 0.6449888605624438,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.41506897571475,
25
+ "eval_loss": 1.1876318454742432,
26
+ "eval_mean_token_accuracy": 0.734131895525511,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.8071,
29
+ "eval_samples_per_second": 15.671,
30
+ "eval_steps_per_second": 1.959,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.049924298375845,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.5795097351074219,
37
+ "learning_rate": 1.9553661870221022e-05,
38
+ "loss": 0.8944448471069336,
39
+ "mean_token_accuracy": 0.7748479396104813,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7996658658565476,
46
+ "eval_loss": 0.7202735543251038,
47
+ "eval_mean_token_accuracy": 0.8070558306089667,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.9199,
50
+ "eval_samples_per_second": 15.831,
51
+ "eval_steps_per_second": 1.979,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7734908878803253,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3136248588562012,
58
+ "learning_rate": 2.9581180778026673e-05,
59
+ "loss": 0.6780608654022217,
60
+ "mean_token_accuracy": 0.8168170280754566,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7119324009778888,
67
+ "eval_loss": 0.6554311513900757,
68
+ "eval_mean_token_accuracy": 0.8215604798738346,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.8692,
71
+ "eval_samples_per_second": 15.84,
72
+ "eval_steps_per_second": 1.98,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7071127541363239,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.387060284614563,
79
+ "learning_rate": 3.960869968583232e-05,
80
+ "loss": 0.6382100582122803,
81
+ "mean_token_accuracy": 0.8229366384446621,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6883931482254073,
88
+ "eval_loss": 0.625065803527832,
89
+ "eval_mean_token_accuracy": 0.828940509710201,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.662,
92
+ "eval_samples_per_second": 15.878,
93
+ "eval_steps_per_second": 1.985,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6800824083387852,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9892916679382324,
100
+ "learning_rate": 4.963621859363797e-05,
101
+ "loss": 0.6011715888977051,
102
+ "mean_token_accuracy": 0.8323964163661003,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6840810470802839,
109
+ "eval_loss": 0.6037028431892395,
110
+ "eval_mean_token_accuracy": 0.8309669033732525,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.4637,
113
+ "eval_samples_per_second": 15.914,
114
+ "eval_steps_per_second": 1.989,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6776216626167297,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.8918434977531433,
121
+ "learning_rate": 5.9663737501443624e-05,
122
+ "loss": 0.5991742610931396,
123
+ "mean_token_accuracy": 0.8300838828086853,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.690427724705186,
130
+ "eval_loss": 0.5939701795578003,
131
+ "eval_mean_token_accuracy": 0.8345950186945671,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.6626,
134
+ "eval_samples_per_second": 15.878,
135
+ "eval_steps_per_second": 1.985,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6709842771291733,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9135531187057495,
142
+ "learning_rate": 6.969125640924927e-05,
143
+ "loss": 0.5914147377014161,
144
+ "mean_token_accuracy": 0.8314545609056949,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6584504666023476,
151
+ "eval_loss": 0.5849721431732178,
152
+ "eval_mean_token_accuracy": 0.8357757236375365,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.3262,
155
+ "eval_samples_per_second": 15.94,
156
+ "eval_steps_per_second": 1.992,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.6524647936224938,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8651587963104248,
163
+ "learning_rate": 7.971877531705493e-05,
164
+ "loss": 0.5710843563079834,
165
+ "mean_token_accuracy": 0.8396127380430698,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6283470298661742,
172
+ "eval_loss": 0.5738973617553711,
173
+ "eval_mean_token_accuracy": 0.8379981181649274,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.5619,
176
+ "eval_samples_per_second": 15.896,
177
+ "eval_steps_per_second": 1.987,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6450445972383022,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8661723732948303,
184
+ "learning_rate": 8.974629422486058e-05,
185
+ "loss": 0.5677794933319091,
186
+ "mean_token_accuracy": 0.8389350369572639,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6142613257086554,
193
+ "eval_loss": 0.5698265433311462,
194
+ "eval_mean_token_accuracy": 0.8388577273418737,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.4443,
197
+ "eval_samples_per_second": 15.918,
198
+ "eval_steps_per_second": 1.99,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6448334597051144,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9662242531776428,
205
+ "learning_rate": 9.977381313266624e-05,
206
+ "loss": 0.581433916091919,
207
+ "mean_token_accuracy": 0.8387043006718159,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6154296522916749,
214
+ "eval_loss": 0.5660303831100464,
215
+ "eval_mean_token_accuracy": 0.8412494766850804,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3063,
218
+ "eval_samples_per_second": 15.943,
219
+ "eval_steps_per_second": 1.993,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6376728117465973,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7618638873100281,
226
+ "learning_rate": 0.00010980133204047189,
227
+ "loss": 0.5678351402282715,
228
+ "mean_token_accuracy": 0.8404546812176704,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6181817033956217,
235
+ "eval_loss": 0.5663750171661377,
236
+ "eval_mean_token_accuracy": 0.8388350962899452,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.5904,
239
+ "eval_samples_per_second": 15.891,
240
+ "eval_steps_per_second": 1.986,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6303176879882812,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.7571695446968079,
247
+ "learning_rate": 0.00011982885094827753,
248
+ "loss": 0.5502778053283691,
249
+ "mean_token_accuracy": 0.8429657347500324,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6252533817707107,
256
+ "eval_loss": 0.5570284128189087,
257
+ "eval_mean_token_accuracy": 0.8427327847064927,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.4157,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.99,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6202544964849949,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.6447190642356873,
268
+ "learning_rate": 0.00012985636985608318,
269
+ "loss": 0.5485352993011474,
270
+ "mean_token_accuracy": 0.844165726006031,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.6441633552312851,
277
+ "eval_loss": 0.5606644153594971,
278
+ "eval_mean_token_accuracy": 0.842403513054515,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.6343,
281
+ "eval_samples_per_second": 15.883,
282
+ "eval_steps_per_second": 1.985,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6306711677461863,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7869907021522522,
289
+ "learning_rate": 0.00013988388876388883,
290
+ "loss": 0.5579307556152344,
291
+ "mean_token_accuracy": 0.841247134655714,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.6263934678809587,
298
+ "eval_loss": 0.5559113025665283,
299
+ "eval_mean_token_accuracy": 0.8427334743183713,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.6403,
302
+ "eval_samples_per_second": 15.882,
303
+ "eval_steps_per_second": 1.985,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6385872110724449,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6679229736328125,
310
+ "learning_rate": 0.0001499114076716945,
311
+ "loss": 0.5667279720306396,
312
+ "mean_token_accuracy": 0.8389136254787445,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6141417321077612,
319
+ "eval_loss": 0.5570600628852844,
320
+ "eval_mean_token_accuracy": 0.8437647996253745,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.7588,
323
+ "eval_samples_per_second": 15.86,
324
+ "eval_steps_per_second": 1.983,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6199494235217571,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.7924400568008423,
331
+ "learning_rate": 0.00015993892657950015,
332
+ "loss": 0.5529299736022949,
333
+ "mean_token_accuracy": 0.8426973208785057,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6133768925833147,
340
+ "eval_loss": 0.556602418422699,
341
+ "eval_mean_token_accuracy": 0.8432947965555413,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.492,
344
+ "eval_samples_per_second": 15.909,
345
+ "eval_steps_per_second": 1.989,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6203986253589392,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8364354372024536,
352
+ "learning_rate": 0.00016996644548730578,
353
+ "loss": 0.5551144123077393,
354
+ "mean_token_accuracy": 0.8432973213493824,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6017442844634833,
361
+ "eval_loss": 0.5566568374633789,
362
+ "eval_mean_token_accuracy": 0.8437666123689607,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.5552,
365
+ "eval_samples_per_second": 15.897,
366
+ "eval_steps_per_second": 1.987,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6341533534228802,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.7783445715904236,
373
+ "learning_rate": 0.00017999396439511144,
374
+ "loss": 0.5669133186340332,
375
+ "mean_token_accuracy": 0.8379446342587471,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6055107958788095,
382
+ "eval_loss": 0.5599350333213806,
383
+ "eval_mean_token_accuracy": 0.8435030894917112,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.4814,
386
+ "eval_samples_per_second": 15.911,
387
+ "eval_steps_per_second": 1.989,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.6306198488920927,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.8449786901473999,
394
+ "learning_rate": 0.0001900214833029171,
395
+ "loss": 0.5739435195922852,
396
+ "mean_token_accuracy": 0.8393832489848136,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6129532439071078,
403
+ "eval_loss": 0.5566295981407166,
404
+ "eval_mean_token_accuracy": 0.8430350880290187,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.4643,
407
+ "eval_samples_per_second": 15.914,
408
+ "eval_steps_per_second": 1.989,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6203123550862074,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 0.7334314584732056,
415
+ "learning_rate": 0.00020004900221072276,
416
+ "loss": 0.5547565937042236,
417
+ "mean_token_accuracy": 0.8403573960065842,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6275761647279873,
424
+ "eval_loss": 0.5621116757392883,
425
+ "eval_mean_token_accuracy": 0.841587379228237,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.4748,
428
+ "eval_samples_per_second": 15.912,
429
+ "eval_steps_per_second": 1.989,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5795013002860241,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 0.8858296871185303,
436
+ "learning_rate": 0.0002015421505577756,
437
+ "loss": 0.5183939933776855,
438
+ "mean_token_accuracy": 0.850081592034071,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5583065545489622,
445
+ "eval_loss": 0.5605642199516296,
446
+ "eval_mean_token_accuracy": 0.8439708411000496,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.5422,
449
+ "eval_samples_per_second": 15.9,
450
+ "eval_steps_per_second": 1.987,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5671238023787737,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.6882498264312744,
457
+ "learning_rate": 0.00020150112347025443,
458
+ "loss": 0.5077326774597168,
459
+ "mean_token_accuracy": 0.8489868573844432,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5868900277933409,
466
+ "eval_loss": 0.5602695345878601,
467
+ "eval_mean_token_accuracy": 0.8428842161977014,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.623,
470
+ "eval_samples_per_second": 15.885,
471
+ "eval_steps_per_second": 1.986,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5533561781048775,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 0.7717723250389099,
478
+ "learning_rate": 0.0002014297192297181,
479
+ "loss": 0.4954517364501953,
480
+ "mean_token_accuracy": 0.8529035650193691,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5600803743961246,
487
+ "eval_loss": 0.5608077645301819,
488
+ "eval_mean_token_accuracy": 0.8445036771685578,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.1316,
491
+ "eval_samples_per_second": 15.976,
492
+ "eval_steps_per_second": 1.997,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5692154694348573,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 0.7322827577590942,
499
+ "learning_rate": 0.0002013279593707117,
500
+ "loss": 0.505049467086792,
501
+ "mean_token_accuracy": 0.8551576808094978,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5732695829383162,
508
+ "eval_loss": 0.5594323873519897,
509
+ "eval_mean_token_accuracy": 0.8449713407560836,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.2726,
512
+ "eval_samples_per_second": 15.949,
513
+ "eval_steps_per_second": 1.994,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5817618492990733,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 1.1776764392852783,
520
+ "learning_rate": 0.0002011958745826208,
521
+ "loss": 0.5137609958648681,
522
+ "mean_token_accuracy": 0.8521522544324398,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5662581343636957,
529
+ "eval_loss": 0.5595026016235352,
530
+ "eval_mean_token_accuracy": 0.8441977164773053,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7261,
533
+ "eval_samples_per_second": 15.866,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5712925456464291,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7960361838340759,
541
+ "learning_rate": 0.0002010335047004159,
542
+ "loss": 0.5134767532348633,
543
+ "mean_token_accuracy": 0.8513577707111836,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.5441222797299541,
550
+ "eval_loss": 0.5535460114479065,
551
+ "eval_mean_token_accuracy": 0.8450886118550633,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.2675,
554
+ "eval_samples_per_second": 15.95,
555
+ "eval_steps_per_second": 1.994,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5787045754492283,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9205410480499268,
562
+ "learning_rate": 0.00020084089869263887,
563
+ "loss": 0.5119701862335205,
564
+ "mean_token_accuracy": 0.8503516331315041,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.5744457827057949,
571
+ "eval_loss": 0.5514978766441345,
572
+ "eval_mean_token_accuracy": 0.845929987901865,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.2299,
575
+ "eval_samples_per_second": 15.957,
576
+ "eval_steps_per_second": 1.995,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5739392962306737,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.7475653886795044,
583
+ "learning_rate": 0.00020061811464663464,
584
+ "loss": 0.5189042091369629,
585
+ "mean_token_accuracy": 0.8492388024926185,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6116398271433142,
592
+ "eval_loss": 0.551732063293457,
593
+ "eval_mean_token_accuracy": 0.8450756967067719,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.6081,
596
+ "eval_samples_per_second": 15.888,
597
+ "eval_steps_per_second": 1.986,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5755622573196888,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.8218411803245544,
604
+ "learning_rate": 0.00020036521975103286,
605
+ "loss": 0.5106248378753662,
606
+ "mean_token_accuracy": 0.8506785586476326,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.5906928708386976,
613
+ "eval_loss": 0.551278829574585,
614
+ "eval_mean_token_accuracy": 0.8462819308042526,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.5438,
617
+ "eval_samples_per_second": 15.899,
618
+ "eval_steps_per_second": 1.987,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5694822132587433,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.8880652189254761,
625
+ "learning_rate": 0.00020008229027548475,
626
+ "loss": 0.5140334606170655,
627
+ "mean_token_accuracy": 0.8521522797644139,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5599641964532608,
634
+ "eval_loss": 0.5501875877380371,
635
+ "eval_mean_token_accuracy": 0.8467660788879838,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6458,
638
+ "eval_samples_per_second": 15.881,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.5675108034163714,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.837087094783783,
646
+ "learning_rate": 0.0001997694115476612,
647
+ "loss": 0.5099846363067627,
648
+ "mean_token_accuracy": 0.8543680295348167,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5728072581249614,
655
+ "eval_loss": 0.5445425510406494,
656
+ "eval_mean_token_accuracy": 0.8474342175001321,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.4859,
659
+ "eval_samples_per_second": 15.91,
660
+ "eval_steps_per_second": 1.989,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5700885068625212,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6598765850067139,
667
+ "learning_rate": 0.000199426677927519,
668
+ "loss": 0.5122694969177246,
669
+ "mean_token_accuracy": 0.8519927568733692,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5476993622128353,
676
+ "eval_loss": 0.5427973866462708,
677
+ "eval_mean_token_accuracy": 0.8478512147138285,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.4172,
680
+ "eval_samples_per_second": 15.923,
681
+ "eval_steps_per_second": 1.99,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.5829229176044464,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.6965194940567017,
688
+ "learning_rate": 0.00019905419277884342,
689
+ "loss": 0.5253659725189209,
690
+ "mean_token_accuracy": 0.8493309423327446,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5666290084983028,
697
+ "eval_loss": 0.5467478036880493,
698
+ "eval_mean_token_accuracy": 0.8479407703460649,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.4414,
701
+ "eval_samples_per_second": 15.918,
702
+ "eval_steps_per_second": 1.99,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5498311135917902,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.636583685874939,
709
+ "learning_rate": 0.00019865206843807482,
710
+ "loss": 0.49981012344360354,
711
+ "mean_token_accuracy": 0.8560848504304885,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.539117265406043,
718
+ "eval_loss": 0.53994220495224,
719
+ "eval_mean_token_accuracy": 0.8488582601380903,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.5296,
722
+ "eval_samples_per_second": 15.902,
723
+ "eval_steps_per_second": 1.988,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5543891470879316,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.6068442463874817,
730
+ "learning_rate": 0.0001982204261804297,
731
+ "loss": 0.498047399520874,
732
+ "mean_token_accuracy": 0.8554679051041603,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5703774151760478,
739
+ "eval_loss": 0.5300245881080627,
740
+ "eval_mean_token_accuracy": 0.850798153946566,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.6456,
743
+ "eval_samples_per_second": 15.881,
744
+ "eval_steps_per_second": 1.985,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.546524541825056,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7274155020713806,
751
+ "learning_rate": 0.00019775939618332566,
752
+ "loss": 0.4988589286804199,
753
+ "mean_token_accuracy": 0.853422473371029,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5614905688305234,
760
+ "eval_loss": 0.5350332260131836,
761
+ "eval_mean_token_accuracy": 0.8492204359797544,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 86.7581,
764
+ "eval_samples_per_second": 15.86,
765
+ "eval_steps_per_second": 1.983,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5519792139530182,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.663466215133667,
772
+ "learning_rate": 0.00019726911748712167,
773
+ "loss": 0.5099314212799072,
774
+ "mean_token_accuracy": 0.848412600159645,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5583519090053647,
781
+ "eval_loss": 0.530483603477478,
782
+ "eval_mean_token_accuracy": 0.8500003374593202,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 86.3961,
785
+ "eval_samples_per_second": 15.927,
786
+ "eval_steps_per_second": 1.991,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.5454779766499996,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.890394926071167,
793
+ "learning_rate": 0.00019674973795318548,
794
+ "loss": 0.4931994915008545,
795
+ "mean_token_accuracy": 0.8540832489728928,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.572755502406941,
802
+ "eval_loss": 0.5415747761726379,
803
+ "eval_mean_token_accuracy": 0.8444425803284312,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4323,
806
+ "eval_samples_per_second": 15.92,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.5392089951783419,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.632411777973175,
814
+ "learning_rate": 0.00019620141421930058,
815
+ "loss": 0.4957888603210449,
816
+ "mean_token_accuracy": 0.8549866065382957,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.540764772961306,
823
+ "eval_loss": 0.5327216386795044,
824
+ "eval_mean_token_accuracy": 0.850631088364956,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.8097,
827
+ "eval_samples_per_second": 15.851,
828
+ "eval_steps_per_second": 1.981,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5674678739160299,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6958843469619751,
835
+ "learning_rate": 0.0001956243116524263,
836
+ "loss": 0.504389762878418,
837
+ "mean_token_accuracy": 0.8527948908507824,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.530262403190136,
844
+ "eval_loss": 0.5308871865272522,
845
+ "eval_mean_token_accuracy": 0.8522498046242913,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.7942,
848
+ "eval_samples_per_second": 15.854,
849
+ "eval_steps_per_second": 1.982,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4742849511213792,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.6941492557525635,
856
+ "learning_rate": 0.00019501860429882556,
857
+ "loss": 0.418599271774292,
858
+ "mean_token_accuracy": 0.8748210859604371,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.504602165069691,
865
+ "eval_loss": 0.542878270149231,
866
+ "eval_mean_token_accuracy": 0.8507604484641275,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.7841,
869
+ "eval_samples_per_second": 15.855,
870
+ "eval_steps_per_second": 1.982,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.45857742577791216,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.5791997909545898,
877
+ "learning_rate": 0.00019438447483157478,
878
+ "loss": 0.399777889251709,
879
+ "mean_token_accuracy": 0.8754058346152306,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.5028848362176918,
886
+ "eval_loss": 0.5356478095054626,
887
+ "eval_mean_token_accuracy": 0.8525635412959165,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.6707,
890
+ "eval_samples_per_second": 15.876,
891
+ "eval_steps_per_second": 1.985,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4869446292519569,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6483516693115234,
898
+ "learning_rate": 0.00019372211449547223,
899
+ "loss": 0.40715818405151366,
900
+ "mean_token_accuracy": 0.875113020837307,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.4928991326759028,
907
+ "eval_loss": 0.5419561862945557,
908
+ "eval_mean_token_accuracy": 0.8516040146350861,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 87.0686,
911
+ "eval_samples_per_second": 15.804,
912
+ "eval_steps_per_second": 1.975,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.45819590501487256,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.6661920547485352,
919
+ "learning_rate": 0.00019303172304936108,
920
+ "loss": 0.39511430263519287,
921
+ "mean_token_accuracy": 0.8780680045485496,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.48602560647698334,
928
+ "eval_loss": 0.5436084866523743,
929
+ "eval_mean_token_accuracy": 0.8500938470973525,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.6809,
932
+ "eval_samples_per_second": 15.874,
933
+ "eval_steps_per_second": 1.984,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4780638810247183,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6870484352111816,
940
+ "learning_rate": 0.0001923135087058851,
941
+ "loss": 0.4061615467071533,
942
+ "mean_token_accuracy": 0.8766494184732437,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.48236206035281337,
949
+ "eval_loss": 0.5446090698242188,
950
+ "eval_mean_token_accuracy": 0.8507725513258646,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.7398,
953
+ "eval_samples_per_second": 15.864,
954
+ "eval_steps_per_second": 1.983,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.463029869645834,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.6894590854644775,
961
+ "learning_rate": 0.00019156768806869427,
962
+ "loss": 0.39602413177490237,
963
+ "mean_token_accuracy": 0.876420046389103,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.4904779093556626,
970
+ "eval_loss": 0.5404934287071228,
971
+ "eval_mean_token_accuracy": 0.852238280828609,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.5348,
974
+ "eval_samples_per_second": 15.901,
975
+ "eval_steps_per_second": 1.988,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4817025110125542,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7756227254867554,
982
+ "learning_rate": 0.00019079448606712033,
983
+ "loss": 0.4177968502044678,
984
+ "mean_token_accuracy": 0.8712256088852882,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5153802815218305,
991
+ "eval_loss": 0.5424937605857849,
992
+ "eval_mean_token_accuracy": 0.8506565759348315,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.8973,
995
+ "eval_samples_per_second": 15.835,
996
+ "eval_steps_per_second": 1.979,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.46456389091908934,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 1.2000319957733154,
1003
+ "learning_rate": 0.00018999413588834105,
1004
+ "loss": 0.4084665775299072,
1005
+ "mean_token_accuracy": 0.8750658087432385,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4849439303195754,
1012
+ "eval_loss": 0.545662522315979,
1013
+ "eval_mean_token_accuracy": 0.8491013112456299,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.9049,
1016
+ "eval_samples_per_second": 15.833,
1017
+ "eval_steps_per_second": 1.979,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.4857471022754908,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.9696341753005981,
1024
+ "learning_rate": 0.0001891668789070541,
1025
+ "loss": 0.4149796962738037,
1026
+ "mean_token_accuracy": 0.8704176343977451,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.4872790058684904,
1033
+ "eval_loss": 0.5412707924842834,
1034
+ "eval_mean_token_accuracy": 0.8509329602468846,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.7846,
1037
+ "eval_samples_per_second": 15.855,
1038
+ "eval_steps_per_second": 1.982,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.4727417893707752,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.7852500677108765,
1045
+ "learning_rate": 0.0001883129646126818,
1046
+ "loss": 0.4142886161804199,
1047
+ "mean_token_accuracy": 0.8712429471313954,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5386548059624295,
1054
+ "eval_loss": 0.536101222038269,
1055
+ "eval_mean_token_accuracy": 0.8499491239009902,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.9501,
1058
+ "eval_samples_per_second": 15.825,
1059
+ "eval_steps_per_second": 1.978,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.4673406321555376,
1064
+ "epoch": 2.5379825653798256,
1065
+ "grad_norm": 0.7133921384811401,
1066
+ "learning_rate": 0.0001874326505341286,
1067
+ "loss": 0.40857529640197754,
1068
+ "mean_token_accuracy": 0.8747925907373428,
1069
+ "num_tokens": 2384270.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.5379825653798256,
1074
+ "eval_entropy": 0.495788364909416,
1075
+ "eval_loss": 0.5418923497200012,
1076
+ "eval_mean_token_accuracy": 0.851321972040243,
1077
+ "eval_num_tokens": 2384270.0,
1078
+ "eval_runtime": 86.7154,
1079
+ "eval_samples_per_second": 15.868,
1080
+ "eval_steps_per_second": 1.983,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.47599745728075504,
1085
+ "epoch": 2.587795765877958,
1086
+ "grad_norm": 0.8202953338623047,
1087
+ "learning_rate": 0.0001865262021621137,
1088
+ "loss": 0.40998234748840334,
1089
+ "mean_token_accuracy": 0.8758242674171924,
1090
+ "num_tokens": 2428036.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.587795765877958,
1095
+ "eval_entropy": 0.4887966953737791,
1096
+ "eval_loss": 0.5408804416656494,
1097
+ "eval_mean_token_accuracy": 0.8512661065473113,
1098
+ "eval_num_tokens": 2428036.0,
1099
+ "eval_runtime": 86.7869,
1100
+ "eval_samples_per_second": 15.855,
1101
+ "eval_steps_per_second": 1.982,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.4824396539479494,
1106
+ "epoch": 2.6376089663760895,
1107
+ "grad_norm": 0.6507360935211182,
1108
+ "learning_rate": 0.00018559389286910275,
1109
+ "loss": 0.4165764808654785,
1110
+ "mean_token_accuracy": 0.8722914069890976,
1111
+ "num_tokens": 2476815.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.6376089663760895,
1116
+ "eval_entropy": 0.4793398808254752,
1117
+ "eval_loss": 0.5326959490776062,
1118
+ "eval_mean_token_accuracy": 0.8534493650807891,
1119
+ "eval_num_tokens": 2476815.0,
1120
+ "eval_runtime": 86.9559,
1121
+ "eval_samples_per_second": 15.824,
1122
+ "eval_steps_per_second": 1.978,
1123
+ "step": 1060
1124
+ },
1125
+ {
1126
+ "entropy": 0.4605010639876127,
1127
+ "epoch": 2.6874221668742218,
1128
+ "grad_norm": 0.6740535497665405,
1129
+ "learning_rate": 0.00018463600382686253,
1130
+ "loss": 0.4123940944671631,
1131
+ "mean_token_accuracy": 0.8733638986945153,
1132
+ "num_tokens": 2527131.0,
1133
+ "step": 1080
1134
+ },
1135
+ {
1136
+ "epoch": 2.6874221668742218,
1137
+ "eval_entropy": 0.47902208583992584,
1138
+ "eval_loss": 0.5372340083122253,
1139
+ "eval_mean_token_accuracy": 0.851325950303743,
1140
+ "eval_num_tokens": 2527131.0,
1141
+ "eval_runtime": 86.9638,
1142
+ "eval_samples_per_second": 15.823,
1143
+ "eval_steps_per_second": 1.978,
1144
+ "step": 1080
1145
+ },
1146
+ {
1147
+ "entropy": 0.4872019402682781,
1148
+ "epoch": 2.7372353673723535,
1149
+ "grad_norm": 0.6994742155075073,
1150
+ "learning_rate": 0.0001836528239216632,
1151
+ "loss": 0.41599602699279786,
1152
+ "mean_token_accuracy": 0.872775862365961,
1153
+ "num_tokens": 2572537.0,
1154
+ "step": 1100
1155
+ },
1156
+ {
1157
+ "epoch": 2.7372353673723535,
1158
+ "eval_entropy": 0.4893243626453156,
1159
+ "eval_loss": 0.5327795743942261,
1160
+ "eval_mean_token_accuracy": 0.8537560302850812,
1161
+ "eval_num_tokens": 2572537.0,
1162
+ "eval_runtime": 86.823,
1163
+ "eval_samples_per_second": 15.848,
1164
+ "eval_steps_per_second": 1.981,
1165
+ "step": 1100
1166
+ },
1167
+ {
1168
+ "entropy": 0.4949610233306885,
1169
+ "epoch": 2.7870485678704857,
1170
+ "grad_norm": 0.9605912566184998,
1171
+ "learning_rate": 0.0001826446496671543,
1172
+ "loss": 0.4266993045806885,
1173
+ "mean_token_accuracy": 0.8688005246222019,
1174
+ "num_tokens": 2616047.0,
1175
+ "step": 1120
1176
+ },
1177
+ {
1178
+ "epoch": 2.7870485678704857,
1179
+ "eval_entropy": 0.5061517927882283,
1180
+ "eval_loss": 0.5356810092926025,
1181
+ "eval_mean_token_accuracy": 0.8524593568818514,
1182
+ "eval_num_tokens": 2616047.0,
1183
+ "eval_runtime": 86.8385,
1184
+ "eval_samples_per_second": 15.846,
1185
+ "eval_steps_per_second": 1.981,
1186
+ "step": 1120
1187
+ }
1188
+ ],
1189
+ "logging_steps": 20,
1190
+ "max_steps": 4020,
1191
+ "num_input_tokens_seen": 0,
1192
+ "num_train_epochs": 10,
1193
+ "save_steps": 20,
1194
+ "stateful_callbacks": {
1195
+ "TrainerControl": {
1196
+ "args": {
1197
+ "should_epoch_stop": false,
1198
+ "should_evaluate": false,
1199
+ "should_log": false,
1200
+ "should_save": true,
1201
+ "should_training_stop": false
1202
+ },
1203
+ "attributes": {}
1204
+ }
1205
+ },
1206
+ "total_flos": 1.1058529480242586e+17,
1207
+ "train_batch_size": 4,
1208
+ "trial_name": null,
1209
+ "trial_params": null
1210
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1140/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1140/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0005183818805460705,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1140/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1140/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1140/trainer_state.json ADDED
@@ -0,0 +1,1231 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.8368617683686175,
6
+ "eval_steps": 20,
7
+ "global_step": 1140,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.9784346982836722,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.0229668617248535,
16
+ "learning_rate": 9.526142962415369e-06,
17
+ "loss": 1.7360023498535155,
18
+ "mean_token_accuracy": 0.6449888605624438,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.41506897571475,
25
+ "eval_loss": 1.1876318454742432,
26
+ "eval_mean_token_accuracy": 0.734131895525511,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.8071,
29
+ "eval_samples_per_second": 15.671,
30
+ "eval_steps_per_second": 1.959,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.049924298375845,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.5795097351074219,
37
+ "learning_rate": 1.9553661870221022e-05,
38
+ "loss": 0.8944448471069336,
39
+ "mean_token_accuracy": 0.7748479396104813,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7996658658565476,
46
+ "eval_loss": 0.7202735543251038,
47
+ "eval_mean_token_accuracy": 0.8070558306089667,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.9199,
50
+ "eval_samples_per_second": 15.831,
51
+ "eval_steps_per_second": 1.979,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7734908878803253,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3136248588562012,
58
+ "learning_rate": 2.9581180778026673e-05,
59
+ "loss": 0.6780608654022217,
60
+ "mean_token_accuracy": 0.8168170280754566,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7119324009778888,
67
+ "eval_loss": 0.6554311513900757,
68
+ "eval_mean_token_accuracy": 0.8215604798738346,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.8692,
71
+ "eval_samples_per_second": 15.84,
72
+ "eval_steps_per_second": 1.98,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7071127541363239,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.387060284614563,
79
+ "learning_rate": 3.960869968583232e-05,
80
+ "loss": 0.6382100582122803,
81
+ "mean_token_accuracy": 0.8229366384446621,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6883931482254073,
88
+ "eval_loss": 0.625065803527832,
89
+ "eval_mean_token_accuracy": 0.828940509710201,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.662,
92
+ "eval_samples_per_second": 15.878,
93
+ "eval_steps_per_second": 1.985,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6800824083387852,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9892916679382324,
100
+ "learning_rate": 4.963621859363797e-05,
101
+ "loss": 0.6011715888977051,
102
+ "mean_token_accuracy": 0.8323964163661003,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6840810470802839,
109
+ "eval_loss": 0.6037028431892395,
110
+ "eval_mean_token_accuracy": 0.8309669033732525,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.4637,
113
+ "eval_samples_per_second": 15.914,
114
+ "eval_steps_per_second": 1.989,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6776216626167297,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.8918434977531433,
121
+ "learning_rate": 5.9663737501443624e-05,
122
+ "loss": 0.5991742610931396,
123
+ "mean_token_accuracy": 0.8300838828086853,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.690427724705186,
130
+ "eval_loss": 0.5939701795578003,
131
+ "eval_mean_token_accuracy": 0.8345950186945671,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.6626,
134
+ "eval_samples_per_second": 15.878,
135
+ "eval_steps_per_second": 1.985,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6709842771291733,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9135531187057495,
142
+ "learning_rate": 6.969125640924927e-05,
143
+ "loss": 0.5914147377014161,
144
+ "mean_token_accuracy": 0.8314545609056949,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6584504666023476,
151
+ "eval_loss": 0.5849721431732178,
152
+ "eval_mean_token_accuracy": 0.8357757236375365,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.3262,
155
+ "eval_samples_per_second": 15.94,
156
+ "eval_steps_per_second": 1.992,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.6524647936224938,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8651587963104248,
163
+ "learning_rate": 7.971877531705493e-05,
164
+ "loss": 0.5710843563079834,
165
+ "mean_token_accuracy": 0.8396127380430698,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6283470298661742,
172
+ "eval_loss": 0.5738973617553711,
173
+ "eval_mean_token_accuracy": 0.8379981181649274,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.5619,
176
+ "eval_samples_per_second": 15.896,
177
+ "eval_steps_per_second": 1.987,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6450445972383022,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8661723732948303,
184
+ "learning_rate": 8.974629422486058e-05,
185
+ "loss": 0.5677794933319091,
186
+ "mean_token_accuracy": 0.8389350369572639,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6142613257086554,
193
+ "eval_loss": 0.5698265433311462,
194
+ "eval_mean_token_accuracy": 0.8388577273418737,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.4443,
197
+ "eval_samples_per_second": 15.918,
198
+ "eval_steps_per_second": 1.99,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6448334597051144,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9662242531776428,
205
+ "learning_rate": 9.977381313266624e-05,
206
+ "loss": 0.581433916091919,
207
+ "mean_token_accuracy": 0.8387043006718159,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6154296522916749,
214
+ "eval_loss": 0.5660303831100464,
215
+ "eval_mean_token_accuracy": 0.8412494766850804,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3063,
218
+ "eval_samples_per_second": 15.943,
219
+ "eval_steps_per_second": 1.993,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6376728117465973,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7618638873100281,
226
+ "learning_rate": 0.00010980133204047189,
227
+ "loss": 0.5678351402282715,
228
+ "mean_token_accuracy": 0.8404546812176704,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6181817033956217,
235
+ "eval_loss": 0.5663750171661377,
236
+ "eval_mean_token_accuracy": 0.8388350962899452,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.5904,
239
+ "eval_samples_per_second": 15.891,
240
+ "eval_steps_per_second": 1.986,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6303176879882812,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.7571695446968079,
247
+ "learning_rate": 0.00011982885094827753,
248
+ "loss": 0.5502778053283691,
249
+ "mean_token_accuracy": 0.8429657347500324,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6252533817707107,
256
+ "eval_loss": 0.5570284128189087,
257
+ "eval_mean_token_accuracy": 0.8427327847064927,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.4157,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.99,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6202544964849949,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.6447190642356873,
268
+ "learning_rate": 0.00012985636985608318,
269
+ "loss": 0.5485352993011474,
270
+ "mean_token_accuracy": 0.844165726006031,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.6441633552312851,
277
+ "eval_loss": 0.5606644153594971,
278
+ "eval_mean_token_accuracy": 0.842403513054515,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.6343,
281
+ "eval_samples_per_second": 15.883,
282
+ "eval_steps_per_second": 1.985,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6306711677461863,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7869907021522522,
289
+ "learning_rate": 0.00013988388876388883,
290
+ "loss": 0.5579307556152344,
291
+ "mean_token_accuracy": 0.841247134655714,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.6263934678809587,
298
+ "eval_loss": 0.5559113025665283,
299
+ "eval_mean_token_accuracy": 0.8427334743183713,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.6403,
302
+ "eval_samples_per_second": 15.882,
303
+ "eval_steps_per_second": 1.985,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6385872110724449,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6679229736328125,
310
+ "learning_rate": 0.0001499114076716945,
311
+ "loss": 0.5667279720306396,
312
+ "mean_token_accuracy": 0.8389136254787445,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6141417321077612,
319
+ "eval_loss": 0.5570600628852844,
320
+ "eval_mean_token_accuracy": 0.8437647996253745,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.7588,
323
+ "eval_samples_per_second": 15.86,
324
+ "eval_steps_per_second": 1.983,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6199494235217571,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.7924400568008423,
331
+ "learning_rate": 0.00015993892657950015,
332
+ "loss": 0.5529299736022949,
333
+ "mean_token_accuracy": 0.8426973208785057,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6133768925833147,
340
+ "eval_loss": 0.556602418422699,
341
+ "eval_mean_token_accuracy": 0.8432947965555413,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.492,
344
+ "eval_samples_per_second": 15.909,
345
+ "eval_steps_per_second": 1.989,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6203986253589392,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8364354372024536,
352
+ "learning_rate": 0.00016996644548730578,
353
+ "loss": 0.5551144123077393,
354
+ "mean_token_accuracy": 0.8432973213493824,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6017442844634833,
361
+ "eval_loss": 0.5566568374633789,
362
+ "eval_mean_token_accuracy": 0.8437666123689607,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.5552,
365
+ "eval_samples_per_second": 15.897,
366
+ "eval_steps_per_second": 1.987,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6341533534228802,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.7783445715904236,
373
+ "learning_rate": 0.00017999396439511144,
374
+ "loss": 0.5669133186340332,
375
+ "mean_token_accuracy": 0.8379446342587471,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6055107958788095,
382
+ "eval_loss": 0.5599350333213806,
383
+ "eval_mean_token_accuracy": 0.8435030894917112,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.4814,
386
+ "eval_samples_per_second": 15.911,
387
+ "eval_steps_per_second": 1.989,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.6306198488920927,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.8449786901473999,
394
+ "learning_rate": 0.0001900214833029171,
395
+ "loss": 0.5739435195922852,
396
+ "mean_token_accuracy": 0.8393832489848136,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6129532439071078,
403
+ "eval_loss": 0.5566295981407166,
404
+ "eval_mean_token_accuracy": 0.8430350880290187,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.4643,
407
+ "eval_samples_per_second": 15.914,
408
+ "eval_steps_per_second": 1.989,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6203123550862074,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 0.7334314584732056,
415
+ "learning_rate": 0.00020004900221072276,
416
+ "loss": 0.5547565937042236,
417
+ "mean_token_accuracy": 0.8403573960065842,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6275761647279873,
424
+ "eval_loss": 0.5621116757392883,
425
+ "eval_mean_token_accuracy": 0.841587379228237,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.4748,
428
+ "eval_samples_per_second": 15.912,
429
+ "eval_steps_per_second": 1.989,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5795013002860241,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 0.8858296871185303,
436
+ "learning_rate": 0.0002015421505577756,
437
+ "loss": 0.5183939933776855,
438
+ "mean_token_accuracy": 0.850081592034071,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5583065545489622,
445
+ "eval_loss": 0.5605642199516296,
446
+ "eval_mean_token_accuracy": 0.8439708411000496,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.5422,
449
+ "eval_samples_per_second": 15.9,
450
+ "eval_steps_per_second": 1.987,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5671238023787737,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.6882498264312744,
457
+ "learning_rate": 0.00020150112347025443,
458
+ "loss": 0.5077326774597168,
459
+ "mean_token_accuracy": 0.8489868573844432,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5868900277933409,
466
+ "eval_loss": 0.5602695345878601,
467
+ "eval_mean_token_accuracy": 0.8428842161977014,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.623,
470
+ "eval_samples_per_second": 15.885,
471
+ "eval_steps_per_second": 1.986,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5533561781048775,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 0.7717723250389099,
478
+ "learning_rate": 0.0002014297192297181,
479
+ "loss": 0.4954517364501953,
480
+ "mean_token_accuracy": 0.8529035650193691,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5600803743961246,
487
+ "eval_loss": 0.5608077645301819,
488
+ "eval_mean_token_accuracy": 0.8445036771685578,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.1316,
491
+ "eval_samples_per_second": 15.976,
492
+ "eval_steps_per_second": 1.997,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5692154694348573,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 0.7322827577590942,
499
+ "learning_rate": 0.0002013279593707117,
500
+ "loss": 0.505049467086792,
501
+ "mean_token_accuracy": 0.8551576808094978,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5732695829383162,
508
+ "eval_loss": 0.5594323873519897,
509
+ "eval_mean_token_accuracy": 0.8449713407560836,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.2726,
512
+ "eval_samples_per_second": 15.949,
513
+ "eval_steps_per_second": 1.994,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5817618492990733,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 1.1776764392852783,
520
+ "learning_rate": 0.0002011958745826208,
521
+ "loss": 0.5137609958648681,
522
+ "mean_token_accuracy": 0.8521522544324398,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5662581343636957,
529
+ "eval_loss": 0.5595026016235352,
530
+ "eval_mean_token_accuracy": 0.8441977164773053,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7261,
533
+ "eval_samples_per_second": 15.866,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5712925456464291,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7960361838340759,
541
+ "learning_rate": 0.0002010335047004159,
542
+ "loss": 0.5134767532348633,
543
+ "mean_token_accuracy": 0.8513577707111836,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.5441222797299541,
550
+ "eval_loss": 0.5535460114479065,
551
+ "eval_mean_token_accuracy": 0.8450886118550633,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.2675,
554
+ "eval_samples_per_second": 15.95,
555
+ "eval_steps_per_second": 1.994,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5787045754492283,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9205410480499268,
562
+ "learning_rate": 0.00020084089869263887,
563
+ "loss": 0.5119701862335205,
564
+ "mean_token_accuracy": 0.8503516331315041,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.5744457827057949,
571
+ "eval_loss": 0.5514978766441345,
572
+ "eval_mean_token_accuracy": 0.845929987901865,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.2299,
575
+ "eval_samples_per_second": 15.957,
576
+ "eval_steps_per_second": 1.995,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5739392962306737,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.7475653886795044,
583
+ "learning_rate": 0.00020061811464663464,
584
+ "loss": 0.5189042091369629,
585
+ "mean_token_accuracy": 0.8492388024926185,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6116398271433142,
592
+ "eval_loss": 0.551732063293457,
593
+ "eval_mean_token_accuracy": 0.8450756967067719,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.6081,
596
+ "eval_samples_per_second": 15.888,
597
+ "eval_steps_per_second": 1.986,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5755622573196888,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.8218411803245544,
604
+ "learning_rate": 0.00020036521975103286,
605
+ "loss": 0.5106248378753662,
606
+ "mean_token_accuracy": 0.8506785586476326,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.5906928708386976,
613
+ "eval_loss": 0.551278829574585,
614
+ "eval_mean_token_accuracy": 0.8462819308042526,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.5438,
617
+ "eval_samples_per_second": 15.899,
618
+ "eval_steps_per_second": 1.987,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5694822132587433,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.8880652189254761,
625
+ "learning_rate": 0.00020008229027548475,
626
+ "loss": 0.5140334606170655,
627
+ "mean_token_accuracy": 0.8521522797644139,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5599641964532608,
634
+ "eval_loss": 0.5501875877380371,
635
+ "eval_mean_token_accuracy": 0.8467660788879838,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6458,
638
+ "eval_samples_per_second": 15.881,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.5675108034163714,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.837087094783783,
646
+ "learning_rate": 0.0001997694115476612,
647
+ "loss": 0.5099846363067627,
648
+ "mean_token_accuracy": 0.8543680295348167,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5728072581249614,
655
+ "eval_loss": 0.5445425510406494,
656
+ "eval_mean_token_accuracy": 0.8474342175001321,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.4859,
659
+ "eval_samples_per_second": 15.91,
660
+ "eval_steps_per_second": 1.989,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5700885068625212,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6598765850067139,
667
+ "learning_rate": 0.000199426677927519,
668
+ "loss": 0.5122694969177246,
669
+ "mean_token_accuracy": 0.8519927568733692,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5476993622128353,
676
+ "eval_loss": 0.5427973866462708,
677
+ "eval_mean_token_accuracy": 0.8478512147138285,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.4172,
680
+ "eval_samples_per_second": 15.923,
681
+ "eval_steps_per_second": 1.99,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.5829229176044464,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.6965194940567017,
688
+ "learning_rate": 0.00019905419277884342,
689
+ "loss": 0.5253659725189209,
690
+ "mean_token_accuracy": 0.8493309423327446,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5666290084983028,
697
+ "eval_loss": 0.5467478036880493,
698
+ "eval_mean_token_accuracy": 0.8479407703460649,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.4414,
701
+ "eval_samples_per_second": 15.918,
702
+ "eval_steps_per_second": 1.99,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5498311135917902,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.636583685874939,
709
+ "learning_rate": 0.00019865206843807482,
710
+ "loss": 0.49981012344360354,
711
+ "mean_token_accuracy": 0.8560848504304885,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.539117265406043,
718
+ "eval_loss": 0.53994220495224,
719
+ "eval_mean_token_accuracy": 0.8488582601380903,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.5296,
722
+ "eval_samples_per_second": 15.902,
723
+ "eval_steps_per_second": 1.988,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5543891470879316,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.6068442463874817,
730
+ "learning_rate": 0.0001982204261804297,
731
+ "loss": 0.498047399520874,
732
+ "mean_token_accuracy": 0.8554679051041603,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5703774151760478,
739
+ "eval_loss": 0.5300245881080627,
740
+ "eval_mean_token_accuracy": 0.850798153946566,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.6456,
743
+ "eval_samples_per_second": 15.881,
744
+ "eval_steps_per_second": 1.985,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.546524541825056,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7274155020713806,
751
+ "learning_rate": 0.00019775939618332566,
752
+ "loss": 0.4988589286804199,
753
+ "mean_token_accuracy": 0.853422473371029,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5614905688305234,
760
+ "eval_loss": 0.5350332260131836,
761
+ "eval_mean_token_accuracy": 0.8492204359797544,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 86.7581,
764
+ "eval_samples_per_second": 15.86,
765
+ "eval_steps_per_second": 1.983,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5519792139530182,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.663466215133667,
772
+ "learning_rate": 0.00019726911748712167,
773
+ "loss": 0.5099314212799072,
774
+ "mean_token_accuracy": 0.848412600159645,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5583519090053647,
781
+ "eval_loss": 0.530483603477478,
782
+ "eval_mean_token_accuracy": 0.8500003374593202,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 86.3961,
785
+ "eval_samples_per_second": 15.927,
786
+ "eval_steps_per_second": 1.991,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.5454779766499996,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.890394926071167,
793
+ "learning_rate": 0.00019674973795318548,
794
+ "loss": 0.4931994915008545,
795
+ "mean_token_accuracy": 0.8540832489728928,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.572755502406941,
802
+ "eval_loss": 0.5415747761726379,
803
+ "eval_mean_token_accuracy": 0.8444425803284312,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4323,
806
+ "eval_samples_per_second": 15.92,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.5392089951783419,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.632411777973175,
814
+ "learning_rate": 0.00019620141421930058,
815
+ "loss": 0.4957888603210449,
816
+ "mean_token_accuracy": 0.8549866065382957,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.540764772961306,
823
+ "eval_loss": 0.5327216386795044,
824
+ "eval_mean_token_accuracy": 0.850631088364956,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.8097,
827
+ "eval_samples_per_second": 15.851,
828
+ "eval_steps_per_second": 1.981,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5674678739160299,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6958843469619751,
835
+ "learning_rate": 0.0001956243116524263,
836
+ "loss": 0.504389762878418,
837
+ "mean_token_accuracy": 0.8527948908507824,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.530262403190136,
844
+ "eval_loss": 0.5308871865272522,
845
+ "eval_mean_token_accuracy": 0.8522498046242913,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.7942,
848
+ "eval_samples_per_second": 15.854,
849
+ "eval_steps_per_second": 1.982,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4742849511213792,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.6941492557525635,
856
+ "learning_rate": 0.00019501860429882556,
857
+ "loss": 0.418599271774292,
858
+ "mean_token_accuracy": 0.8748210859604371,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.504602165069691,
865
+ "eval_loss": 0.542878270149231,
866
+ "eval_mean_token_accuracy": 0.8507604484641275,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.7841,
869
+ "eval_samples_per_second": 15.855,
870
+ "eval_steps_per_second": 1.982,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.45857742577791216,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.5791997909545898,
877
+ "learning_rate": 0.00019438447483157478,
878
+ "loss": 0.399777889251709,
879
+ "mean_token_accuracy": 0.8754058346152306,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.5028848362176918,
886
+ "eval_loss": 0.5356478095054626,
887
+ "eval_mean_token_accuracy": 0.8525635412959165,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.6707,
890
+ "eval_samples_per_second": 15.876,
891
+ "eval_steps_per_second": 1.985,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4869446292519569,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6483516693115234,
898
+ "learning_rate": 0.00019372211449547223,
899
+ "loss": 0.40715818405151366,
900
+ "mean_token_accuracy": 0.875113020837307,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.4928991326759028,
907
+ "eval_loss": 0.5419561862945557,
908
+ "eval_mean_token_accuracy": 0.8516040146350861,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 87.0686,
911
+ "eval_samples_per_second": 15.804,
912
+ "eval_steps_per_second": 1.975,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.45819590501487256,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.6661920547485352,
919
+ "learning_rate": 0.00019303172304936108,
920
+ "loss": 0.39511430263519287,
921
+ "mean_token_accuracy": 0.8780680045485496,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.48602560647698334,
928
+ "eval_loss": 0.5436084866523743,
929
+ "eval_mean_token_accuracy": 0.8500938470973525,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.6809,
932
+ "eval_samples_per_second": 15.874,
933
+ "eval_steps_per_second": 1.984,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4780638810247183,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6870484352111816,
940
+ "learning_rate": 0.0001923135087058851,
941
+ "loss": 0.4061615467071533,
942
+ "mean_token_accuracy": 0.8766494184732437,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.48236206035281337,
949
+ "eval_loss": 0.5446090698242188,
950
+ "eval_mean_token_accuracy": 0.8507725513258646,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.7398,
953
+ "eval_samples_per_second": 15.864,
954
+ "eval_steps_per_second": 1.983,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.463029869645834,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.6894590854644775,
961
+ "learning_rate": 0.00019156768806869427,
962
+ "loss": 0.39602413177490237,
963
+ "mean_token_accuracy": 0.876420046389103,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.4904779093556626,
970
+ "eval_loss": 0.5404934287071228,
971
+ "eval_mean_token_accuracy": 0.852238280828609,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.5348,
974
+ "eval_samples_per_second": 15.901,
975
+ "eval_steps_per_second": 1.988,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4817025110125542,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7756227254867554,
982
+ "learning_rate": 0.00019079448606712033,
983
+ "loss": 0.4177968502044678,
984
+ "mean_token_accuracy": 0.8712256088852882,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5153802815218305,
991
+ "eval_loss": 0.5424937605857849,
992
+ "eval_mean_token_accuracy": 0.8506565759348315,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.8973,
995
+ "eval_samples_per_second": 15.835,
996
+ "eval_steps_per_second": 1.979,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.46456389091908934,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 1.2000319957733154,
1003
+ "learning_rate": 0.00018999413588834105,
1004
+ "loss": 0.4084665775299072,
1005
+ "mean_token_accuracy": 0.8750658087432385,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4849439303195754,
1012
+ "eval_loss": 0.545662522315979,
1013
+ "eval_mean_token_accuracy": 0.8491013112456299,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.9049,
1016
+ "eval_samples_per_second": 15.833,
1017
+ "eval_steps_per_second": 1.979,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.4857471022754908,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.9696341753005981,
1024
+ "learning_rate": 0.0001891668789070541,
1025
+ "loss": 0.4149796962738037,
1026
+ "mean_token_accuracy": 0.8704176343977451,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.4872790058684904,
1033
+ "eval_loss": 0.5412707924842834,
1034
+ "eval_mean_token_accuracy": 0.8509329602468846,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.7846,
1037
+ "eval_samples_per_second": 15.855,
1038
+ "eval_steps_per_second": 1.982,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.4727417893707752,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.7852500677108765,
1045
+ "learning_rate": 0.0001883129646126818,
1046
+ "loss": 0.4142886161804199,
1047
+ "mean_token_accuracy": 0.8712429471313954,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5386548059624295,
1054
+ "eval_loss": 0.536101222038269,
1055
+ "eval_mean_token_accuracy": 0.8499491239009902,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.9501,
1058
+ "eval_samples_per_second": 15.825,
1059
+ "eval_steps_per_second": 1.978,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.4673406321555376,
1064
+ "epoch": 2.5379825653798256,
1065
+ "grad_norm": 0.7133921384811401,
1066
+ "learning_rate": 0.0001874326505341286,
1067
+ "loss": 0.40857529640197754,
1068
+ "mean_token_accuracy": 0.8747925907373428,
1069
+ "num_tokens": 2384270.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.5379825653798256,
1074
+ "eval_entropy": 0.495788364909416,
1075
+ "eval_loss": 0.5418923497200012,
1076
+ "eval_mean_token_accuracy": 0.851321972040243,
1077
+ "eval_num_tokens": 2384270.0,
1078
+ "eval_runtime": 86.7154,
1079
+ "eval_samples_per_second": 15.868,
1080
+ "eval_steps_per_second": 1.983,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.47599745728075504,
1085
+ "epoch": 2.587795765877958,
1086
+ "grad_norm": 0.8202953338623047,
1087
+ "learning_rate": 0.0001865262021621137,
1088
+ "loss": 0.40998234748840334,
1089
+ "mean_token_accuracy": 0.8758242674171924,
1090
+ "num_tokens": 2428036.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.587795765877958,
1095
+ "eval_entropy": 0.4887966953737791,
1096
+ "eval_loss": 0.5408804416656494,
1097
+ "eval_mean_token_accuracy": 0.8512661065473113,
1098
+ "eval_num_tokens": 2428036.0,
1099
+ "eval_runtime": 86.7869,
1100
+ "eval_samples_per_second": 15.855,
1101
+ "eval_steps_per_second": 1.982,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.4824396539479494,
1106
+ "epoch": 2.6376089663760895,
1107
+ "grad_norm": 0.6507360935211182,
1108
+ "learning_rate": 0.00018559389286910275,
1109
+ "loss": 0.4165764808654785,
1110
+ "mean_token_accuracy": 0.8722914069890976,
1111
+ "num_tokens": 2476815.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.6376089663760895,
1116
+ "eval_entropy": 0.4793398808254752,
1117
+ "eval_loss": 0.5326959490776062,
1118
+ "eval_mean_token_accuracy": 0.8534493650807891,
1119
+ "eval_num_tokens": 2476815.0,
1120
+ "eval_runtime": 86.9559,
1121
+ "eval_samples_per_second": 15.824,
1122
+ "eval_steps_per_second": 1.978,
1123
+ "step": 1060
1124
+ },
1125
+ {
1126
+ "entropy": 0.4605010639876127,
1127
+ "epoch": 2.6874221668742218,
1128
+ "grad_norm": 0.6740535497665405,
1129
+ "learning_rate": 0.00018463600382686253,
1130
+ "loss": 0.4123940944671631,
1131
+ "mean_token_accuracy": 0.8733638986945153,
1132
+ "num_tokens": 2527131.0,
1133
+ "step": 1080
1134
+ },
1135
+ {
1136
+ "epoch": 2.6874221668742218,
1137
+ "eval_entropy": 0.47902208583992584,
1138
+ "eval_loss": 0.5372340083122253,
1139
+ "eval_mean_token_accuracy": 0.851325950303743,
1140
+ "eval_num_tokens": 2527131.0,
1141
+ "eval_runtime": 86.9638,
1142
+ "eval_samples_per_second": 15.823,
1143
+ "eval_steps_per_second": 1.978,
1144
+ "step": 1080
1145
+ },
1146
+ {
1147
+ "entropy": 0.4872019402682781,
1148
+ "epoch": 2.7372353673723535,
1149
+ "grad_norm": 0.6994742155075073,
1150
+ "learning_rate": 0.0001836528239216632,
1151
+ "loss": 0.41599602699279786,
1152
+ "mean_token_accuracy": 0.872775862365961,
1153
+ "num_tokens": 2572537.0,
1154
+ "step": 1100
1155
+ },
1156
+ {
1157
+ "epoch": 2.7372353673723535,
1158
+ "eval_entropy": 0.4893243626453156,
1159
+ "eval_loss": 0.5327795743942261,
1160
+ "eval_mean_token_accuracy": 0.8537560302850812,
1161
+ "eval_num_tokens": 2572537.0,
1162
+ "eval_runtime": 86.823,
1163
+ "eval_samples_per_second": 15.848,
1164
+ "eval_steps_per_second": 1.981,
1165
+ "step": 1100
1166
+ },
1167
+ {
1168
+ "entropy": 0.4949610233306885,
1169
+ "epoch": 2.7870485678704857,
1170
+ "grad_norm": 0.9605912566184998,
1171
+ "learning_rate": 0.0001826446496671543,
1172
+ "loss": 0.4266993045806885,
1173
+ "mean_token_accuracy": 0.8688005246222019,
1174
+ "num_tokens": 2616047.0,
1175
+ "step": 1120
1176
+ },
1177
+ {
1178
+ "epoch": 2.7870485678704857,
1179
+ "eval_entropy": 0.5061517927882283,
1180
+ "eval_loss": 0.5356810092926025,
1181
+ "eval_mean_token_accuracy": 0.8524593568818514,
1182
+ "eval_num_tokens": 2616047.0,
1183
+ "eval_runtime": 86.8385,
1184
+ "eval_samples_per_second": 15.846,
1185
+ "eval_steps_per_second": 1.981,
1186
+ "step": 1120
1187
+ },
1188
+ {
1189
+ "entropy": 0.48401356525719164,
1190
+ "epoch": 2.8368617683686175,
1191
+ "grad_norm": 0.6332499980926514,
1192
+ "learning_rate": 0.00018161178511494022,
1193
+ "loss": 0.42131738662719725,
1194
+ "mean_token_accuracy": 0.8729447312653065,
1195
+ "num_tokens": 2664744.0,
1196
+ "step": 1140
1197
+ },
1198
+ {
1199
+ "epoch": 2.8368617683686175,
1200
+ "eval_entropy": 0.48792897060860035,
1201
+ "eval_loss": 0.5290402173995972,
1202
+ "eval_mean_token_accuracy": 0.853078076659247,
1203
+ "eval_num_tokens": 2664744.0,
1204
+ "eval_runtime": 86.7746,
1205
+ "eval_samples_per_second": 15.857,
1206
+ "eval_steps_per_second": 1.982,
1207
+ "step": 1140
1208
+ }
1209
+ ],
1210
+ "logging_steps": 20,
1211
+ "max_steps": 4020,
1212
+ "num_input_tokens_seen": 0,
1213
+ "num_train_epochs": 10,
1214
+ "save_steps": 20,
1215
+ "stateful_callbacks": {
1216
+ "TrainerControl": {
1217
+ "args": {
1218
+ "should_epoch_stop": false,
1219
+ "should_evaluate": false,
1220
+ "should_log": false,
1221
+ "should_save": true,
1222
+ "should_training_stop": false
1223
+ },
1224
+ "attributes": {}
1225
+ }
1226
+ },
1227
+ "total_flos": 1.1277422592347136e+17,
1228
+ "train_batch_size": 4,
1229
+ "trial_name": null,
1230
+ "trial_params": null
1231
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1160/README.md ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen3.5-4B-Base
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen3.5-4B-Base
7
+ - lora
8
+ - sft
9
+ - transformers
10
+ - trl
11
+ ---
12
+
13
+ # Model Card for Model ID
14
+
15
+ <!-- Provide a quick summary of what the model is/does. -->
16
+
17
+
18
+
19
+ ## Model Details
20
+
21
+ ### Model Description
22
+
23
+ <!-- Provide a longer summary of what this model is. -->
24
+
25
+
26
+
27
+ - **Developed by:** [More Information Needed]
28
+ - **Funded by [optional]:** [More Information Needed]
29
+ - **Shared by [optional]:** [More Information Needed]
30
+ - **Model type:** [More Information Needed]
31
+ - **Language(s) (NLP):** [More Information Needed]
32
+ - **License:** [More Information Needed]
33
+ - **Finetuned from model [optional]:** [More Information Needed]
34
+
35
+ ### Model Sources [optional]
36
+
37
+ <!-- Provide the basic links for the model. -->
38
+
39
+ - **Repository:** [More Information Needed]
40
+ - **Paper [optional]:** [More Information Needed]
41
+ - **Demo [optional]:** [More Information Needed]
42
+
43
+ ## Uses
44
+
45
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
46
+
47
+ ### Direct Use
48
+
49
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
50
+
51
+ [More Information Needed]
52
+
53
+ ### Downstream Use [optional]
54
+
55
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
56
+
57
+ [More Information Needed]
58
+
59
+ ### Out-of-Scope Use
60
+
61
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
62
+
63
+ [More Information Needed]
64
+
65
+ ## Bias, Risks, and Limitations
66
+
67
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
68
+
69
+ [More Information Needed]
70
+
71
+ ### Recommendations
72
+
73
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
74
+
75
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
76
+
77
+ ## How to Get Started with the Model
78
+
79
+ Use the code below to get started with the model.
80
+
81
+ [More Information Needed]
82
+
83
+ ## Training Details
84
+
85
+ ### Training Data
86
+
87
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
88
+
89
+ [More Information Needed]
90
+
91
+ ### Training Procedure
92
+
93
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
94
+
95
+ #### Preprocessing [optional]
96
+
97
+ [More Information Needed]
98
+
99
+
100
+ #### Training Hyperparameters
101
+
102
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
103
+
104
+ #### Speeds, Sizes, Times [optional]
105
+
106
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
107
+
108
+ [More Information Needed]
109
+
110
+ ## Evaluation
111
+
112
+ <!-- This section describes the evaluation protocols and provides the results. -->
113
+
114
+ ### Testing Data, Factors & Metrics
115
+
116
+ #### Testing Data
117
+
118
+ <!-- This should link to a Dataset Card if possible. -->
119
+
120
+ [More Information Needed]
121
+
122
+ #### Factors
123
+
124
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
125
+
126
+ [More Information Needed]
127
+
128
+ #### Metrics
129
+
130
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
131
+
132
+ [More Information Needed]
133
+
134
+ ### Results
135
+
136
+ [More Information Needed]
137
+
138
+ #### Summary
139
+
140
+
141
+
142
+ ## Model Examination [optional]
143
+
144
+ <!-- Relevant interpretability work for the model goes here -->
145
+
146
+ [More Information Needed]
147
+
148
+ ## Environmental Impact
149
+
150
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
151
+
152
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
153
+
154
+ - **Hardware Type:** [More Information Needed]
155
+ - **Hours used:** [More Information Needed]
156
+ - **Cloud Provider:** [More Information Needed]
157
+ - **Compute Region:** [More Information Needed]
158
+ - **Carbon Emitted:** [More Information Needed]
159
+
160
+ ## Technical Specifications [optional]
161
+
162
+ ### Model Architecture and Objective
163
+
164
+ [More Information Needed]
165
+
166
+ ### Compute Infrastructure
167
+
168
+ [More Information Needed]
169
+
170
+ #### Hardware
171
+
172
+ [More Information Needed]
173
+
174
+ #### Software
175
+
176
+ [More Information Needed]
177
+
178
+ ## Citation [optional]
179
+
180
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
181
+
182
+ **BibTeX:**
183
+
184
+ [More Information Needed]
185
+
186
+ **APA:**
187
+
188
+ [More Information Needed]
189
+
190
+ ## Glossary [optional]
191
+
192
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
193
+
194
+ [More Information Needed]
195
+
196
+ ## More Information [optional]
197
+
198
+ [More Information Needed]
199
+
200
+ ## Model Card Authors [optional]
201
+
202
+ [More Information Needed]
203
+
204
+ ## Model Card Contact
205
+
206
+ [More Information Needed]
207
+ ### Framework versions
208
+
209
+ - PEFT 0.18.1
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1160/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-4B-Base",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0005183818805460705,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "q_proj",
34
+ "o_proj",
35
+ "v_proj",
36
+ "k_proj",
37
+ "gate_proj",
38
+ "down_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1160/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1160/tokenizer_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|endoftext|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "model_max_length": 262144,
14
+ "model_specific_special_tokens": {
15
+ "audio_bos_token": "<|audio_start|>",
16
+ "audio_eos_token": "<|audio_end|>",
17
+ "audio_token": "<|audio_pad|>",
18
+ "image_token": "<|image_pad|>",
19
+ "video_token": "<|video_pad|>",
20
+ "vision_bos_token": "<|vision_start|>",
21
+ "vision_eos_token": "<|vision_end|>"
22
+ },
23
+ "pad_token": "<|endoftext|>",
24
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
25
+ "split_special_tokens": false,
26
+ "tokenizer_class": "TokenizersBackend",
27
+ "unk_token": null,
28
+ "video_token": "<|video_pad|>",
29
+ "vision_bos_token": "<|vision_start|>",
30
+ "vision_eos_token": "<|vision_end|>"
31
+ }
overgeneralisation_original_Swedish/Qwen3.5-4B-Base_overgeneralisation_splits_original_features_train_overgeneralisation_splits_original_features_test1/checkpoint-1160/trainer_state.json ADDED
@@ -0,0 +1,1252 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.8866749688667497,
6
+ "eval_steps": 20,
7
+ "global_step": 1160,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.9784346982836722,
14
+ "epoch": 0.049813200498132,
15
+ "grad_norm": 3.0229668617248535,
16
+ "learning_rate": 9.526142962415369e-06,
17
+ "loss": 1.7360023498535155,
18
+ "mean_token_accuracy": 0.6449888605624438,
19
+ "num_tokens": 46794.0,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.049813200498132,
24
+ "eval_entropy": 1.41506897571475,
25
+ "eval_loss": 1.1876318454742432,
26
+ "eval_mean_token_accuracy": 0.734131895525511,
27
+ "eval_num_tokens": 46794.0,
28
+ "eval_runtime": 87.8071,
29
+ "eval_samples_per_second": 15.671,
30
+ "eval_steps_per_second": 1.959,
31
+ "step": 20
32
+ },
33
+ {
34
+ "entropy": 1.049924298375845,
35
+ "epoch": 0.099626400996264,
36
+ "grad_norm": 1.5795097351074219,
37
+ "learning_rate": 1.9553661870221022e-05,
38
+ "loss": 0.8944448471069336,
39
+ "mean_token_accuracy": 0.7748479396104813,
40
+ "num_tokens": 90754.0,
41
+ "step": 40
42
+ },
43
+ {
44
+ "epoch": 0.099626400996264,
45
+ "eval_entropy": 0.7996658658565476,
46
+ "eval_loss": 0.7202735543251038,
47
+ "eval_mean_token_accuracy": 0.8070558306089667,
48
+ "eval_num_tokens": 90754.0,
49
+ "eval_runtime": 86.9199,
50
+ "eval_samples_per_second": 15.831,
51
+ "eval_steps_per_second": 1.979,
52
+ "step": 40
53
+ },
54
+ {
55
+ "entropy": 0.7734908878803253,
56
+ "epoch": 0.149439601494396,
57
+ "grad_norm": 1.3136248588562012,
58
+ "learning_rate": 2.9581180778026673e-05,
59
+ "loss": 0.6780608654022217,
60
+ "mean_token_accuracy": 0.8168170280754566,
61
+ "num_tokens": 137472.0,
62
+ "step": 60
63
+ },
64
+ {
65
+ "epoch": 0.149439601494396,
66
+ "eval_entropy": 0.7119324009778888,
67
+ "eval_loss": 0.6554311513900757,
68
+ "eval_mean_token_accuracy": 0.8215604798738346,
69
+ "eval_num_tokens": 137472.0,
70
+ "eval_runtime": 86.8692,
71
+ "eval_samples_per_second": 15.84,
72
+ "eval_steps_per_second": 1.98,
73
+ "step": 60
74
+ },
75
+ {
76
+ "entropy": 0.7071127541363239,
77
+ "epoch": 0.199252801992528,
78
+ "grad_norm": 1.387060284614563,
79
+ "learning_rate": 3.960869968583232e-05,
80
+ "loss": 0.6382100582122803,
81
+ "mean_token_accuracy": 0.8229366384446621,
82
+ "num_tokens": 187408.0,
83
+ "step": 80
84
+ },
85
+ {
86
+ "epoch": 0.199252801992528,
87
+ "eval_entropy": 0.6883931482254073,
88
+ "eval_loss": 0.625065803527832,
89
+ "eval_mean_token_accuracy": 0.828940509710201,
90
+ "eval_num_tokens": 187408.0,
91
+ "eval_runtime": 86.662,
92
+ "eval_samples_per_second": 15.878,
93
+ "eval_steps_per_second": 1.985,
94
+ "step": 80
95
+ },
96
+ {
97
+ "entropy": 0.6800824083387852,
98
+ "epoch": 0.24906600249066002,
99
+ "grad_norm": 0.9892916679382324,
100
+ "learning_rate": 4.963621859363797e-05,
101
+ "loss": 0.6011715888977051,
102
+ "mean_token_accuracy": 0.8323964163661003,
103
+ "num_tokens": 234197.0,
104
+ "step": 100
105
+ },
106
+ {
107
+ "epoch": 0.24906600249066002,
108
+ "eval_entropy": 0.6840810470802839,
109
+ "eval_loss": 0.6037028431892395,
110
+ "eval_mean_token_accuracy": 0.8309669033732525,
111
+ "eval_num_tokens": 234197.0,
112
+ "eval_runtime": 86.4637,
113
+ "eval_samples_per_second": 15.914,
114
+ "eval_steps_per_second": 1.989,
115
+ "step": 100
116
+ },
117
+ {
118
+ "entropy": 0.6776216626167297,
119
+ "epoch": 0.298879202988792,
120
+ "grad_norm": 0.8918434977531433,
121
+ "learning_rate": 5.9663737501443624e-05,
122
+ "loss": 0.5991742610931396,
123
+ "mean_token_accuracy": 0.8300838828086853,
124
+ "num_tokens": 281241.0,
125
+ "step": 120
126
+ },
127
+ {
128
+ "epoch": 0.298879202988792,
129
+ "eval_entropy": 0.690427724705186,
130
+ "eval_loss": 0.5939701795578003,
131
+ "eval_mean_token_accuracy": 0.8345950186945671,
132
+ "eval_num_tokens": 281241.0,
133
+ "eval_runtime": 86.6626,
134
+ "eval_samples_per_second": 15.878,
135
+ "eval_steps_per_second": 1.985,
136
+ "step": 120
137
+ },
138
+ {
139
+ "entropy": 0.6709842771291733,
140
+ "epoch": 0.34869240348692404,
141
+ "grad_norm": 0.9135531187057495,
142
+ "learning_rate": 6.969125640924927e-05,
143
+ "loss": 0.5914147377014161,
144
+ "mean_token_accuracy": 0.8314545609056949,
145
+ "num_tokens": 327393.0,
146
+ "step": 140
147
+ },
148
+ {
149
+ "epoch": 0.34869240348692404,
150
+ "eval_entropy": 0.6584504666023476,
151
+ "eval_loss": 0.5849721431732178,
152
+ "eval_mean_token_accuracy": 0.8357757236375365,
153
+ "eval_num_tokens": 327393.0,
154
+ "eval_runtime": 86.3262,
155
+ "eval_samples_per_second": 15.94,
156
+ "eval_steps_per_second": 1.992,
157
+ "step": 140
158
+ },
159
+ {
160
+ "entropy": 0.6524647936224938,
161
+ "epoch": 0.398505603985056,
162
+ "grad_norm": 0.8651587963104248,
163
+ "learning_rate": 7.971877531705493e-05,
164
+ "loss": 0.5710843563079834,
165
+ "mean_token_accuracy": 0.8396127380430698,
166
+ "num_tokens": 373834.0,
167
+ "step": 160
168
+ },
169
+ {
170
+ "epoch": 0.398505603985056,
171
+ "eval_entropy": 0.6283470298661742,
172
+ "eval_loss": 0.5738973617553711,
173
+ "eval_mean_token_accuracy": 0.8379981181649274,
174
+ "eval_num_tokens": 373834.0,
175
+ "eval_runtime": 86.5619,
176
+ "eval_samples_per_second": 15.896,
177
+ "eval_steps_per_second": 1.987,
178
+ "step": 160
179
+ },
180
+ {
181
+ "entropy": 0.6450445972383022,
182
+ "epoch": 0.44831880448318806,
183
+ "grad_norm": 0.8661723732948303,
184
+ "learning_rate": 8.974629422486058e-05,
185
+ "loss": 0.5677794933319091,
186
+ "mean_token_accuracy": 0.8389350369572639,
187
+ "num_tokens": 422572.0,
188
+ "step": 180
189
+ },
190
+ {
191
+ "epoch": 0.44831880448318806,
192
+ "eval_entropy": 0.6142613257086554,
193
+ "eval_loss": 0.5698265433311462,
194
+ "eval_mean_token_accuracy": 0.8388577273418737,
195
+ "eval_num_tokens": 422572.0,
196
+ "eval_runtime": 86.4443,
197
+ "eval_samples_per_second": 15.918,
198
+ "eval_steps_per_second": 1.99,
199
+ "step": 180
200
+ },
201
+ {
202
+ "entropy": 0.6448334597051144,
203
+ "epoch": 0.49813200498132004,
204
+ "grad_norm": 0.9662242531776428,
205
+ "learning_rate": 9.977381313266624e-05,
206
+ "loss": 0.581433916091919,
207
+ "mean_token_accuracy": 0.8387043006718159,
208
+ "num_tokens": 471879.0,
209
+ "step": 200
210
+ },
211
+ {
212
+ "epoch": 0.49813200498132004,
213
+ "eval_entropy": 0.6154296522916749,
214
+ "eval_loss": 0.5660303831100464,
215
+ "eval_mean_token_accuracy": 0.8412494766850804,
216
+ "eval_num_tokens": 471879.0,
217
+ "eval_runtime": 86.3063,
218
+ "eval_samples_per_second": 15.943,
219
+ "eval_steps_per_second": 1.993,
220
+ "step": 200
221
+ },
222
+ {
223
+ "entropy": 0.6376728117465973,
224
+ "epoch": 0.547945205479452,
225
+ "grad_norm": 0.7618638873100281,
226
+ "learning_rate": 0.00010980133204047189,
227
+ "loss": 0.5678351402282715,
228
+ "mean_token_accuracy": 0.8404546812176704,
229
+ "num_tokens": 520984.0,
230
+ "step": 220
231
+ },
232
+ {
233
+ "epoch": 0.547945205479452,
234
+ "eval_entropy": 0.6181817033956217,
235
+ "eval_loss": 0.5663750171661377,
236
+ "eval_mean_token_accuracy": 0.8388350962899452,
237
+ "eval_num_tokens": 520984.0,
238
+ "eval_runtime": 86.5904,
239
+ "eval_samples_per_second": 15.891,
240
+ "eval_steps_per_second": 1.986,
241
+ "step": 220
242
+ },
243
+ {
244
+ "entropy": 0.6303176879882812,
245
+ "epoch": 0.597758405977584,
246
+ "grad_norm": 0.7571695446968079,
247
+ "learning_rate": 0.00011982885094827753,
248
+ "loss": 0.5502778053283691,
249
+ "mean_token_accuracy": 0.8429657347500324,
250
+ "num_tokens": 566596.0,
251
+ "step": 240
252
+ },
253
+ {
254
+ "epoch": 0.597758405977584,
255
+ "eval_entropy": 0.6252533817707107,
256
+ "eval_loss": 0.5570284128189087,
257
+ "eval_mean_token_accuracy": 0.8427327847064927,
258
+ "eval_num_tokens": 566596.0,
259
+ "eval_runtime": 86.4157,
260
+ "eval_samples_per_second": 15.923,
261
+ "eval_steps_per_second": 1.99,
262
+ "step": 240
263
+ },
264
+ {
265
+ "entropy": 0.6202544964849949,
266
+ "epoch": 0.6475716064757161,
267
+ "grad_norm": 0.6447190642356873,
268
+ "learning_rate": 0.00012985636985608318,
269
+ "loss": 0.5485352993011474,
270
+ "mean_token_accuracy": 0.844165726006031,
271
+ "num_tokens": 613603.0,
272
+ "step": 260
273
+ },
274
+ {
275
+ "epoch": 0.6475716064757161,
276
+ "eval_entropy": 0.6441633552312851,
277
+ "eval_loss": 0.5606644153594971,
278
+ "eval_mean_token_accuracy": 0.842403513054515,
279
+ "eval_num_tokens": 613603.0,
280
+ "eval_runtime": 86.6343,
281
+ "eval_samples_per_second": 15.883,
282
+ "eval_steps_per_second": 1.985,
283
+ "step": 260
284
+ },
285
+ {
286
+ "entropy": 0.6306711677461863,
287
+ "epoch": 0.6973848069738481,
288
+ "grad_norm": 0.7869907021522522,
289
+ "learning_rate": 0.00013988388876388883,
290
+ "loss": 0.5579307556152344,
291
+ "mean_token_accuracy": 0.841247134655714,
292
+ "num_tokens": 658565.0,
293
+ "step": 280
294
+ },
295
+ {
296
+ "epoch": 0.6973848069738481,
297
+ "eval_entropy": 0.6263934678809587,
298
+ "eval_loss": 0.5559113025665283,
299
+ "eval_mean_token_accuracy": 0.8427334743183713,
300
+ "eval_num_tokens": 658565.0,
301
+ "eval_runtime": 86.6403,
302
+ "eval_samples_per_second": 15.882,
303
+ "eval_steps_per_second": 1.985,
304
+ "step": 280
305
+ },
306
+ {
307
+ "entropy": 0.6385872110724449,
308
+ "epoch": 0.7471980074719801,
309
+ "grad_norm": 0.6679229736328125,
310
+ "learning_rate": 0.0001499114076716945,
311
+ "loss": 0.5667279720306396,
312
+ "mean_token_accuracy": 0.8389136254787445,
313
+ "num_tokens": 705680.0,
314
+ "step": 300
315
+ },
316
+ {
317
+ "epoch": 0.7471980074719801,
318
+ "eval_entropy": 0.6141417321077612,
319
+ "eval_loss": 0.5570600628852844,
320
+ "eval_mean_token_accuracy": 0.8437647996253745,
321
+ "eval_num_tokens": 705680.0,
322
+ "eval_runtime": 86.7588,
323
+ "eval_samples_per_second": 15.86,
324
+ "eval_steps_per_second": 1.983,
325
+ "step": 300
326
+ },
327
+ {
328
+ "entropy": 0.6199494235217571,
329
+ "epoch": 0.797011207970112,
330
+ "grad_norm": 0.7924400568008423,
331
+ "learning_rate": 0.00015993892657950015,
332
+ "loss": 0.5529299736022949,
333
+ "mean_token_accuracy": 0.8426973208785057,
334
+ "num_tokens": 752616.0,
335
+ "step": 320
336
+ },
337
+ {
338
+ "epoch": 0.797011207970112,
339
+ "eval_entropy": 0.6133768925833147,
340
+ "eval_loss": 0.556602418422699,
341
+ "eval_mean_token_accuracy": 0.8432947965555413,
342
+ "eval_num_tokens": 752616.0,
343
+ "eval_runtime": 86.492,
344
+ "eval_samples_per_second": 15.909,
345
+ "eval_steps_per_second": 1.989,
346
+ "step": 320
347
+ },
348
+ {
349
+ "entropy": 0.6203986253589392,
350
+ "epoch": 0.8468244084682441,
351
+ "grad_norm": 0.8364354372024536,
352
+ "learning_rate": 0.00016996644548730578,
353
+ "loss": 0.5551144123077393,
354
+ "mean_token_accuracy": 0.8432973213493824,
355
+ "num_tokens": 797151.0,
356
+ "step": 340
357
+ },
358
+ {
359
+ "epoch": 0.8468244084682441,
360
+ "eval_entropy": 0.6017442844634833,
361
+ "eval_loss": 0.5566568374633789,
362
+ "eval_mean_token_accuracy": 0.8437666123689607,
363
+ "eval_num_tokens": 797151.0,
364
+ "eval_runtime": 86.5552,
365
+ "eval_samples_per_second": 15.897,
366
+ "eval_steps_per_second": 1.987,
367
+ "step": 340
368
+ },
369
+ {
370
+ "entropy": 0.6341533534228802,
371
+ "epoch": 0.8966376089663761,
372
+ "grad_norm": 0.7783445715904236,
373
+ "learning_rate": 0.00017999396439511144,
374
+ "loss": 0.5669133186340332,
375
+ "mean_token_accuracy": 0.8379446342587471,
376
+ "num_tokens": 843585.0,
377
+ "step": 360
378
+ },
379
+ {
380
+ "epoch": 0.8966376089663761,
381
+ "eval_entropy": 0.6055107958788095,
382
+ "eval_loss": 0.5599350333213806,
383
+ "eval_mean_token_accuracy": 0.8435030894917112,
384
+ "eval_num_tokens": 843585.0,
385
+ "eval_runtime": 86.4814,
386
+ "eval_samples_per_second": 15.911,
387
+ "eval_steps_per_second": 1.989,
388
+ "step": 360
389
+ },
390
+ {
391
+ "entropy": 0.6306198488920927,
392
+ "epoch": 0.9464508094645081,
393
+ "grad_norm": 0.8449786901473999,
394
+ "learning_rate": 0.0001900214833029171,
395
+ "loss": 0.5739435195922852,
396
+ "mean_token_accuracy": 0.8393832489848136,
397
+ "num_tokens": 889842.0,
398
+ "step": 380
399
+ },
400
+ {
401
+ "epoch": 0.9464508094645081,
402
+ "eval_entropy": 0.6129532439071078,
403
+ "eval_loss": 0.5566295981407166,
404
+ "eval_mean_token_accuracy": 0.8430350880290187,
405
+ "eval_num_tokens": 889842.0,
406
+ "eval_runtime": 86.4643,
407
+ "eval_samples_per_second": 15.914,
408
+ "eval_steps_per_second": 1.989,
409
+ "step": 380
410
+ },
411
+ {
412
+ "entropy": 0.6203123550862074,
413
+ "epoch": 0.9962640099626401,
414
+ "grad_norm": 0.7334314584732056,
415
+ "learning_rate": 0.00020004900221072276,
416
+ "loss": 0.5547565937042236,
417
+ "mean_token_accuracy": 0.8403573960065842,
418
+ "num_tokens": 935589.0,
419
+ "step": 400
420
+ },
421
+ {
422
+ "epoch": 0.9962640099626401,
423
+ "eval_entropy": 0.6275761647279873,
424
+ "eval_loss": 0.5621116757392883,
425
+ "eval_mean_token_accuracy": 0.841587379228237,
426
+ "eval_num_tokens": 935589.0,
427
+ "eval_runtime": 86.4748,
428
+ "eval_samples_per_second": 15.912,
429
+ "eval_steps_per_second": 1.989,
430
+ "step": 400
431
+ },
432
+ {
433
+ "entropy": 0.5795013002860241,
434
+ "epoch": 1.0448318804483188,
435
+ "grad_norm": 0.8858296871185303,
436
+ "learning_rate": 0.0002015421505577756,
437
+ "loss": 0.5183939933776855,
438
+ "mean_token_accuracy": 0.850081592034071,
439
+ "num_tokens": 980589.0,
440
+ "step": 420
441
+ },
442
+ {
443
+ "epoch": 1.0448318804483188,
444
+ "eval_entropy": 0.5583065545489622,
445
+ "eval_loss": 0.5605642199516296,
446
+ "eval_mean_token_accuracy": 0.8439708411000496,
447
+ "eval_num_tokens": 980589.0,
448
+ "eval_runtime": 86.5422,
449
+ "eval_samples_per_second": 15.9,
450
+ "eval_steps_per_second": 1.987,
451
+ "step": 420
452
+ },
453
+ {
454
+ "entropy": 0.5671238023787737,
455
+ "epoch": 1.0946450809464507,
456
+ "grad_norm": 0.6882498264312744,
457
+ "learning_rate": 0.00020150112347025443,
458
+ "loss": 0.5077326774597168,
459
+ "mean_token_accuracy": 0.8489868573844432,
460
+ "num_tokens": 1027852.0,
461
+ "step": 440
462
+ },
463
+ {
464
+ "epoch": 1.0946450809464507,
465
+ "eval_entropy": 0.5868900277933409,
466
+ "eval_loss": 0.5602695345878601,
467
+ "eval_mean_token_accuracy": 0.8428842161977014,
468
+ "eval_num_tokens": 1027852.0,
469
+ "eval_runtime": 86.623,
470
+ "eval_samples_per_second": 15.885,
471
+ "eval_steps_per_second": 1.986,
472
+ "step": 440
473
+ },
474
+ {
475
+ "entropy": 0.5533561781048775,
476
+ "epoch": 1.1444582814445827,
477
+ "grad_norm": 0.7717723250389099,
478
+ "learning_rate": 0.0002014297192297181,
479
+ "loss": 0.4954517364501953,
480
+ "mean_token_accuracy": 0.8529035650193691,
481
+ "num_tokens": 1077649.0,
482
+ "step": 460
483
+ },
484
+ {
485
+ "epoch": 1.1444582814445827,
486
+ "eval_entropy": 0.5600803743961246,
487
+ "eval_loss": 0.5608077645301819,
488
+ "eval_mean_token_accuracy": 0.8445036771685578,
489
+ "eval_num_tokens": 1077649.0,
490
+ "eval_runtime": 86.1316,
491
+ "eval_samples_per_second": 15.976,
492
+ "eval_steps_per_second": 1.997,
493
+ "step": 460
494
+ },
495
+ {
496
+ "entropy": 0.5692154694348573,
497
+ "epoch": 1.1942714819427147,
498
+ "grad_norm": 0.7322827577590942,
499
+ "learning_rate": 0.0002013279593707117,
500
+ "loss": 0.505049467086792,
501
+ "mean_token_accuracy": 0.8551576808094978,
502
+ "num_tokens": 1124872.0,
503
+ "step": 480
504
+ },
505
+ {
506
+ "epoch": 1.1942714819427147,
507
+ "eval_entropy": 0.5732695829383162,
508
+ "eval_loss": 0.5594323873519897,
509
+ "eval_mean_token_accuracy": 0.8449713407560836,
510
+ "eval_num_tokens": 1124872.0,
511
+ "eval_runtime": 86.2726,
512
+ "eval_samples_per_second": 15.949,
513
+ "eval_steps_per_second": 1.994,
514
+ "step": 480
515
+ },
516
+ {
517
+ "entropy": 0.5817618492990733,
518
+ "epoch": 1.244084682440847,
519
+ "grad_norm": 1.1776764392852783,
520
+ "learning_rate": 0.0002011958745826208,
521
+ "loss": 0.5137609958648681,
522
+ "mean_token_accuracy": 0.8521522544324398,
523
+ "num_tokens": 1168698.0,
524
+ "step": 500
525
+ },
526
+ {
527
+ "epoch": 1.244084682440847,
528
+ "eval_entropy": 0.5662581343636957,
529
+ "eval_loss": 0.5595026016235352,
530
+ "eval_mean_token_accuracy": 0.8441977164773053,
531
+ "eval_num_tokens": 1168698.0,
532
+ "eval_runtime": 86.7261,
533
+ "eval_samples_per_second": 15.866,
534
+ "eval_steps_per_second": 1.983,
535
+ "step": 500
536
+ },
537
+ {
538
+ "entropy": 0.5712925456464291,
539
+ "epoch": 1.293897882938979,
540
+ "grad_norm": 0.7960361838340759,
541
+ "learning_rate": 0.0002010335047004159,
542
+ "loss": 0.5134767532348633,
543
+ "mean_token_accuracy": 0.8513577707111836,
544
+ "num_tokens": 1216679.0,
545
+ "step": 520
546
+ },
547
+ {
548
+ "epoch": 1.293897882938979,
549
+ "eval_entropy": 0.5441222797299541,
550
+ "eval_loss": 0.5535460114479065,
551
+ "eval_mean_token_accuracy": 0.8450886118550633,
552
+ "eval_num_tokens": 1216679.0,
553
+ "eval_runtime": 86.2675,
554
+ "eval_samples_per_second": 15.95,
555
+ "eval_steps_per_second": 1.994,
556
+ "step": 520
557
+ },
558
+ {
559
+ "entropy": 0.5787045754492283,
560
+ "epoch": 1.3437110834371109,
561
+ "grad_norm": 0.9205410480499268,
562
+ "learning_rate": 0.00020084089869263887,
563
+ "loss": 0.5119701862335205,
564
+ "mean_token_accuracy": 0.8503516331315041,
565
+ "num_tokens": 1261365.0,
566
+ "step": 540
567
+ },
568
+ {
569
+ "epoch": 1.3437110834371109,
570
+ "eval_entropy": 0.5744457827057949,
571
+ "eval_loss": 0.5514978766441345,
572
+ "eval_mean_token_accuracy": 0.845929987901865,
573
+ "eval_num_tokens": 1261365.0,
574
+ "eval_runtime": 86.2299,
575
+ "eval_samples_per_second": 15.957,
576
+ "eval_steps_per_second": 1.995,
577
+ "step": 540
578
+ },
579
+ {
580
+ "entropy": 0.5739392962306737,
581
+ "epoch": 1.3935242839352429,
582
+ "grad_norm": 0.7475653886795044,
583
+ "learning_rate": 0.00020061811464663464,
584
+ "loss": 0.5189042091369629,
585
+ "mean_token_accuracy": 0.8492388024926185,
586
+ "num_tokens": 1306879.0,
587
+ "step": 560
588
+ },
589
+ {
590
+ "epoch": 1.3935242839352429,
591
+ "eval_entropy": 0.6116398271433142,
592
+ "eval_loss": 0.551732063293457,
593
+ "eval_mean_token_accuracy": 0.8450756967067719,
594
+ "eval_num_tokens": 1306879.0,
595
+ "eval_runtime": 86.6081,
596
+ "eval_samples_per_second": 15.888,
597
+ "eval_steps_per_second": 1.986,
598
+ "step": 560
599
+ },
600
+ {
601
+ "entropy": 0.5755622573196888,
602
+ "epoch": 1.4433374844333748,
603
+ "grad_norm": 0.8218411803245544,
604
+ "learning_rate": 0.00020036521975103286,
605
+ "loss": 0.5106248378753662,
606
+ "mean_token_accuracy": 0.8506785586476326,
607
+ "num_tokens": 1353534.0,
608
+ "step": 580
609
+ },
610
+ {
611
+ "epoch": 1.4433374844333748,
612
+ "eval_entropy": 0.5906928708386976,
613
+ "eval_loss": 0.551278829574585,
614
+ "eval_mean_token_accuracy": 0.8462819308042526,
615
+ "eval_num_tokens": 1353534.0,
616
+ "eval_runtime": 86.5438,
617
+ "eval_samples_per_second": 15.899,
618
+ "eval_steps_per_second": 1.987,
619
+ "step": 580
620
+ },
621
+ {
622
+ "entropy": 0.5694822132587433,
623
+ "epoch": 1.4931506849315068,
624
+ "grad_norm": 0.8880652189254761,
625
+ "learning_rate": 0.00020008229027548475,
626
+ "loss": 0.5140334606170655,
627
+ "mean_token_accuracy": 0.8521522797644139,
628
+ "num_tokens": 1399537.0,
629
+ "step": 600
630
+ },
631
+ {
632
+ "epoch": 1.4931506849315068,
633
+ "eval_entropy": 0.5599641964532608,
634
+ "eval_loss": 0.5501875877380371,
635
+ "eval_mean_token_accuracy": 0.8467660788879838,
636
+ "eval_num_tokens": 1399537.0,
637
+ "eval_runtime": 86.6458,
638
+ "eval_samples_per_second": 15.881,
639
+ "eval_steps_per_second": 1.985,
640
+ "step": 600
641
+ },
642
+ {
643
+ "entropy": 0.5675108034163714,
644
+ "epoch": 1.5429638854296388,
645
+ "grad_norm": 0.837087094783783,
646
+ "learning_rate": 0.0001997694115476612,
647
+ "loss": 0.5099846363067627,
648
+ "mean_token_accuracy": 0.8543680295348167,
649
+ "num_tokens": 1448422.0,
650
+ "step": 620
651
+ },
652
+ {
653
+ "epoch": 1.5429638854296388,
654
+ "eval_entropy": 0.5728072581249614,
655
+ "eval_loss": 0.5445425510406494,
656
+ "eval_mean_token_accuracy": 0.8474342175001321,
657
+ "eval_num_tokens": 1448422.0,
658
+ "eval_runtime": 86.4859,
659
+ "eval_samples_per_second": 15.91,
660
+ "eval_steps_per_second": 1.989,
661
+ "step": 620
662
+ },
663
+ {
664
+ "entropy": 0.5700885068625212,
665
+ "epoch": 1.592777085927771,
666
+ "grad_norm": 0.6598765850067139,
667
+ "learning_rate": 0.000199426677927519,
668
+ "loss": 0.5122694969177246,
669
+ "mean_token_accuracy": 0.8519927568733692,
670
+ "num_tokens": 1495009.0,
671
+ "step": 640
672
+ },
673
+ {
674
+ "epoch": 1.592777085927771,
675
+ "eval_entropy": 0.5476993622128353,
676
+ "eval_loss": 0.5427973866462708,
677
+ "eval_mean_token_accuracy": 0.8478512147138285,
678
+ "eval_num_tokens": 1495009.0,
679
+ "eval_runtime": 86.4172,
680
+ "eval_samples_per_second": 15.923,
681
+ "eval_steps_per_second": 1.99,
682
+ "step": 640
683
+ },
684
+ {
685
+ "entropy": 0.5829229176044464,
686
+ "epoch": 1.6425902864259028,
687
+ "grad_norm": 0.6965194940567017,
688
+ "learning_rate": 0.00019905419277884342,
689
+ "loss": 0.5253659725189209,
690
+ "mean_token_accuracy": 0.8493309423327446,
691
+ "num_tokens": 1536932.0,
692
+ "step": 660
693
+ },
694
+ {
695
+ "epoch": 1.6425902864259028,
696
+ "eval_entropy": 0.5666290084983028,
697
+ "eval_loss": 0.5467478036880493,
698
+ "eval_mean_token_accuracy": 0.8479407703460649,
699
+ "eval_num_tokens": 1536932.0,
700
+ "eval_runtime": 86.4414,
701
+ "eval_samples_per_second": 15.918,
702
+ "eval_steps_per_second": 1.99,
703
+ "step": 660
704
+ },
705
+ {
706
+ "entropy": 0.5498311135917902,
707
+ "epoch": 1.692403486924035,
708
+ "grad_norm": 0.636583685874939,
709
+ "learning_rate": 0.00019865206843807482,
710
+ "loss": 0.49981012344360354,
711
+ "mean_token_accuracy": 0.8560848504304885,
712
+ "num_tokens": 1585718.0,
713
+ "step": 680
714
+ },
715
+ {
716
+ "epoch": 1.692403486924035,
717
+ "eval_entropy": 0.539117265406043,
718
+ "eval_loss": 0.53994220495224,
719
+ "eval_mean_token_accuracy": 0.8488582601380903,
720
+ "eval_num_tokens": 1585718.0,
721
+ "eval_runtime": 86.5296,
722
+ "eval_samples_per_second": 15.902,
723
+ "eval_steps_per_second": 1.988,
724
+ "step": 680
725
+ },
726
+ {
727
+ "entropy": 0.5543891470879316,
728
+ "epoch": 1.7422166874221667,
729
+ "grad_norm": 0.6068442463874817,
730
+ "learning_rate": 0.0001982204261804297,
731
+ "loss": 0.498047399520874,
732
+ "mean_token_accuracy": 0.8554679051041603,
733
+ "num_tokens": 1635718.0,
734
+ "step": 700
735
+ },
736
+ {
737
+ "epoch": 1.7422166874221667,
738
+ "eval_entropy": 0.5703774151760478,
739
+ "eval_loss": 0.5300245881080627,
740
+ "eval_mean_token_accuracy": 0.850798153946566,
741
+ "eval_num_tokens": 1635718.0,
742
+ "eval_runtime": 86.6456,
743
+ "eval_samples_per_second": 15.881,
744
+ "eval_steps_per_second": 1.985,
745
+ "step": 700
746
+ },
747
+ {
748
+ "entropy": 0.546524541825056,
749
+ "epoch": 1.792029887920299,
750
+ "grad_norm": 0.7274155020713806,
751
+ "learning_rate": 0.00019775939618332566,
752
+ "loss": 0.4988589286804199,
753
+ "mean_token_accuracy": 0.853422473371029,
754
+ "num_tokens": 1681291.0,
755
+ "step": 720
756
+ },
757
+ {
758
+ "epoch": 1.792029887920299,
759
+ "eval_entropy": 0.5614905688305234,
760
+ "eval_loss": 0.5350332260131836,
761
+ "eval_mean_token_accuracy": 0.8492204359797544,
762
+ "eval_num_tokens": 1681291.0,
763
+ "eval_runtime": 86.7581,
764
+ "eval_samples_per_second": 15.86,
765
+ "eval_steps_per_second": 1.983,
766
+ "step": 720
767
+ },
768
+ {
769
+ "entropy": 0.5519792139530182,
770
+ "epoch": 1.841843088418431,
771
+ "grad_norm": 0.663466215133667,
772
+ "learning_rate": 0.00019726911748712167,
773
+ "loss": 0.5099314212799072,
774
+ "mean_token_accuracy": 0.848412600159645,
775
+ "num_tokens": 1729102.0,
776
+ "step": 740
777
+ },
778
+ {
779
+ "epoch": 1.841843088418431,
780
+ "eval_entropy": 0.5583519090053647,
781
+ "eval_loss": 0.530483603477478,
782
+ "eval_mean_token_accuracy": 0.8500003374593202,
783
+ "eval_num_tokens": 1729102.0,
784
+ "eval_runtime": 86.3961,
785
+ "eval_samples_per_second": 15.927,
786
+ "eval_steps_per_second": 1.991,
787
+ "step": 740
788
+ },
789
+ {
790
+ "entropy": 0.5454779766499996,
791
+ "epoch": 1.891656288916563,
792
+ "grad_norm": 0.890394926071167,
793
+ "learning_rate": 0.00019674973795318548,
794
+ "loss": 0.4931994915008545,
795
+ "mean_token_accuracy": 0.8540832489728928,
796
+ "num_tokens": 1773578.0,
797
+ "step": 760
798
+ },
799
+ {
800
+ "epoch": 1.891656288916563,
801
+ "eval_entropy": 0.572755502406941,
802
+ "eval_loss": 0.5415747761726379,
803
+ "eval_mean_token_accuracy": 0.8444425803284312,
804
+ "eval_num_tokens": 1773578.0,
805
+ "eval_runtime": 86.4323,
806
+ "eval_samples_per_second": 15.92,
807
+ "eval_steps_per_second": 1.99,
808
+ "step": 760
809
+ },
810
+ {
811
+ "entropy": 0.5392089951783419,
812
+ "epoch": 1.9414694894146949,
813
+ "grad_norm": 0.632411777973175,
814
+ "learning_rate": 0.00019620141421930058,
815
+ "loss": 0.4957888603210449,
816
+ "mean_token_accuracy": 0.8549866065382957,
817
+ "num_tokens": 1821725.0,
818
+ "step": 780
819
+ },
820
+ {
821
+ "epoch": 1.9414694894146949,
822
+ "eval_entropy": 0.540764772961306,
823
+ "eval_loss": 0.5327216386795044,
824
+ "eval_mean_token_accuracy": 0.850631088364956,
825
+ "eval_num_tokens": 1821725.0,
826
+ "eval_runtime": 86.8097,
827
+ "eval_samples_per_second": 15.851,
828
+ "eval_steps_per_second": 1.981,
829
+ "step": 780
830
+ },
831
+ {
832
+ "entropy": 0.5674678739160299,
833
+ "epoch": 1.9912826899128269,
834
+ "grad_norm": 0.6958843469619751,
835
+ "learning_rate": 0.0001956243116524263,
836
+ "loss": 0.504389762878418,
837
+ "mean_token_accuracy": 0.8527948908507824,
838
+ "num_tokens": 1868431.0,
839
+ "step": 800
840
+ },
841
+ {
842
+ "epoch": 1.9912826899128269,
843
+ "eval_entropy": 0.530262403190136,
844
+ "eval_loss": 0.5308871865272522,
845
+ "eval_mean_token_accuracy": 0.8522498046242913,
846
+ "eval_num_tokens": 1868431.0,
847
+ "eval_runtime": 86.7942,
848
+ "eval_samples_per_second": 15.854,
849
+ "eval_steps_per_second": 1.982,
850
+ "step": 800
851
+ },
852
+ {
853
+ "entropy": 0.4742849511213792,
854
+ "epoch": 2.0398505603985058,
855
+ "grad_norm": 0.6941492557525635,
856
+ "learning_rate": 0.00019501860429882556,
857
+ "loss": 0.418599271774292,
858
+ "mean_token_accuracy": 0.8748210859604371,
859
+ "num_tokens": 1915280.0,
860
+ "step": 820
861
+ },
862
+ {
863
+ "epoch": 2.0398505603985058,
864
+ "eval_entropy": 0.504602165069691,
865
+ "eval_loss": 0.542878270149231,
866
+ "eval_mean_token_accuracy": 0.8507604484641275,
867
+ "eval_num_tokens": 1915280.0,
868
+ "eval_runtime": 86.7841,
869
+ "eval_samples_per_second": 15.855,
870
+ "eval_steps_per_second": 1.982,
871
+ "step": 820
872
+ },
873
+ {
874
+ "entropy": 0.45857742577791216,
875
+ "epoch": 2.0896637608966375,
876
+ "grad_norm": 0.5791997909545898,
877
+ "learning_rate": 0.00019438447483157478,
878
+ "loss": 0.399777889251709,
879
+ "mean_token_accuracy": 0.8754058346152306,
880
+ "num_tokens": 1965306.0,
881
+ "step": 840
882
+ },
883
+ {
884
+ "epoch": 2.0896637608966375,
885
+ "eval_entropy": 0.5028848362176918,
886
+ "eval_loss": 0.5356478095054626,
887
+ "eval_mean_token_accuracy": 0.8525635412959165,
888
+ "eval_num_tokens": 1965306.0,
889
+ "eval_runtime": 86.6707,
890
+ "eval_samples_per_second": 15.876,
891
+ "eval_steps_per_second": 1.985,
892
+ "step": 840
893
+ },
894
+ {
895
+ "entropy": 0.4869446292519569,
896
+ "epoch": 2.1394769613947697,
897
+ "grad_norm": 0.6483516693115234,
898
+ "learning_rate": 0.00019372211449547223,
899
+ "loss": 0.40715818405151366,
900
+ "mean_token_accuracy": 0.875113020837307,
901
+ "num_tokens": 2008562.0,
902
+ "step": 860
903
+ },
904
+ {
905
+ "epoch": 2.1394769613947697,
906
+ "eval_entropy": 0.4928991326759028,
907
+ "eval_loss": 0.5419561862945557,
908
+ "eval_mean_token_accuracy": 0.8516040146350861,
909
+ "eval_num_tokens": 2008562.0,
910
+ "eval_runtime": 87.0686,
911
+ "eval_samples_per_second": 15.804,
912
+ "eval_steps_per_second": 1.975,
913
+ "step": 860
914
+ },
915
+ {
916
+ "entropy": 0.45819590501487256,
917
+ "epoch": 2.1892901618929015,
918
+ "grad_norm": 0.6661920547485352,
919
+ "learning_rate": 0.00019303172304936108,
920
+ "loss": 0.39511430263519287,
921
+ "mean_token_accuracy": 0.8780680045485496,
922
+ "num_tokens": 2056474.0,
923
+ "step": 880
924
+ },
925
+ {
926
+ "epoch": 2.1892901618929015,
927
+ "eval_entropy": 0.48602560647698334,
928
+ "eval_loss": 0.5436084866523743,
929
+ "eval_mean_token_accuracy": 0.8500938470973525,
930
+ "eval_num_tokens": 2056474.0,
931
+ "eval_runtime": 86.6809,
932
+ "eval_samples_per_second": 15.874,
933
+ "eval_steps_per_second": 1.984,
934
+ "step": 880
935
+ },
936
+ {
937
+ "entropy": 0.4780638810247183,
938
+ "epoch": 2.2391033623910337,
939
+ "grad_norm": 0.6870484352111816,
940
+ "learning_rate": 0.0001923135087058851,
941
+ "loss": 0.4061615467071533,
942
+ "mean_token_accuracy": 0.8766494184732437,
943
+ "num_tokens": 2103543.0,
944
+ "step": 900
945
+ },
946
+ {
947
+ "epoch": 2.2391033623910337,
948
+ "eval_entropy": 0.48236206035281337,
949
+ "eval_loss": 0.5446090698242188,
950
+ "eval_mean_token_accuracy": 0.8507725513258646,
951
+ "eval_num_tokens": 2103543.0,
952
+ "eval_runtime": 86.7398,
953
+ "eval_samples_per_second": 15.864,
954
+ "eval_steps_per_second": 1.983,
955
+ "step": 900
956
+ },
957
+ {
958
+ "entropy": 0.463029869645834,
959
+ "epoch": 2.2889165628891655,
960
+ "grad_norm": 0.6894590854644775,
961
+ "learning_rate": 0.00019156768806869427,
962
+ "loss": 0.39602413177490237,
963
+ "mean_token_accuracy": 0.876420046389103,
964
+ "num_tokens": 2147861.0,
965
+ "step": 920
966
+ },
967
+ {
968
+ "epoch": 2.2889165628891655,
969
+ "eval_entropy": 0.4904779093556626,
970
+ "eval_loss": 0.5404934287071228,
971
+ "eval_mean_token_accuracy": 0.852238280828609,
972
+ "eval_num_tokens": 2147861.0,
973
+ "eval_runtime": 86.5348,
974
+ "eval_samples_per_second": 15.901,
975
+ "eval_steps_per_second": 1.988,
976
+ "step": 920
977
+ },
978
+ {
979
+ "entropy": 0.4817025110125542,
980
+ "epoch": 2.3387297633872977,
981
+ "grad_norm": 0.7756227254867554,
982
+ "learning_rate": 0.00019079448606712033,
983
+ "loss": 0.4177968502044678,
984
+ "mean_token_accuracy": 0.8712256088852882,
985
+ "num_tokens": 2190561.0,
986
+ "step": 940
987
+ },
988
+ {
989
+ "epoch": 2.3387297633872977,
990
+ "eval_entropy": 0.5153802815218305,
991
+ "eval_loss": 0.5424937605857849,
992
+ "eval_mean_token_accuracy": 0.8506565759348315,
993
+ "eval_num_tokens": 2190561.0,
994
+ "eval_runtime": 86.8973,
995
+ "eval_samples_per_second": 15.835,
996
+ "eval_steps_per_second": 1.979,
997
+ "step": 940
998
+ },
999
+ {
1000
+ "entropy": 0.46456389091908934,
1001
+ "epoch": 2.3885429638854294,
1002
+ "grad_norm": 1.2000319957733154,
1003
+ "learning_rate": 0.00018999413588834105,
1004
+ "loss": 0.4084665775299072,
1005
+ "mean_token_accuracy": 0.8750658087432385,
1006
+ "num_tokens": 2239412.0,
1007
+ "step": 960
1008
+ },
1009
+ {
1010
+ "epoch": 2.3885429638854294,
1011
+ "eval_entropy": 0.4849439303195754,
1012
+ "eval_loss": 0.545662522315979,
1013
+ "eval_mean_token_accuracy": 0.8491013112456299,
1014
+ "eval_num_tokens": 2239412.0,
1015
+ "eval_runtime": 86.9049,
1016
+ "eval_samples_per_second": 15.833,
1017
+ "eval_steps_per_second": 1.979,
1018
+ "step": 960
1019
+ },
1020
+ {
1021
+ "entropy": 0.4857471022754908,
1022
+ "epoch": 2.4383561643835616,
1023
+ "grad_norm": 0.9696341753005981,
1024
+ "learning_rate": 0.0001891668789070541,
1025
+ "loss": 0.4149796962738037,
1026
+ "mean_token_accuracy": 0.8704176343977451,
1027
+ "num_tokens": 2286283.0,
1028
+ "step": 980
1029
+ },
1030
+ {
1031
+ "epoch": 2.4383561643835616,
1032
+ "eval_entropy": 0.4872790058684904,
1033
+ "eval_loss": 0.5412707924842834,
1034
+ "eval_mean_token_accuracy": 0.8509329602468846,
1035
+ "eval_num_tokens": 2286283.0,
1036
+ "eval_runtime": 86.7846,
1037
+ "eval_samples_per_second": 15.855,
1038
+ "eval_steps_per_second": 1.982,
1039
+ "step": 980
1040
+ },
1041
+ {
1042
+ "entropy": 0.4727417893707752,
1043
+ "epoch": 2.488169364881694,
1044
+ "grad_norm": 0.7852500677108765,
1045
+ "learning_rate": 0.0001883129646126818,
1046
+ "loss": 0.4142886161804199,
1047
+ "mean_token_accuracy": 0.8712429471313954,
1048
+ "num_tokens": 2333733.0,
1049
+ "step": 1000
1050
+ },
1051
+ {
1052
+ "epoch": 2.488169364881694,
1053
+ "eval_entropy": 0.5386548059624295,
1054
+ "eval_loss": 0.536101222038269,
1055
+ "eval_mean_token_accuracy": 0.8499491239009902,
1056
+ "eval_num_tokens": 2333733.0,
1057
+ "eval_runtime": 86.9501,
1058
+ "eval_samples_per_second": 15.825,
1059
+ "eval_steps_per_second": 1.978,
1060
+ "step": 1000
1061
+ },
1062
+ {
1063
+ "entropy": 0.4673406321555376,
1064
+ "epoch": 2.5379825653798256,
1065
+ "grad_norm": 0.7133921384811401,
1066
+ "learning_rate": 0.0001874326505341286,
1067
+ "loss": 0.40857529640197754,
1068
+ "mean_token_accuracy": 0.8747925907373428,
1069
+ "num_tokens": 2384270.0,
1070
+ "step": 1020
1071
+ },
1072
+ {
1073
+ "epoch": 2.5379825653798256,
1074
+ "eval_entropy": 0.495788364909416,
1075
+ "eval_loss": 0.5418923497200012,
1076
+ "eval_mean_token_accuracy": 0.851321972040243,
1077
+ "eval_num_tokens": 2384270.0,
1078
+ "eval_runtime": 86.7154,
1079
+ "eval_samples_per_second": 15.868,
1080
+ "eval_steps_per_second": 1.983,
1081
+ "step": 1020
1082
+ },
1083
+ {
1084
+ "entropy": 0.47599745728075504,
1085
+ "epoch": 2.587795765877958,
1086
+ "grad_norm": 0.8202953338623047,
1087
+ "learning_rate": 0.0001865262021621137,
1088
+ "loss": 0.40998234748840334,
1089
+ "mean_token_accuracy": 0.8758242674171924,
1090
+ "num_tokens": 2428036.0,
1091
+ "step": 1040
1092
+ },
1093
+ {
1094
+ "epoch": 2.587795765877958,
1095
+ "eval_entropy": 0.4887966953737791,
1096
+ "eval_loss": 0.5408804416656494,
1097
+ "eval_mean_token_accuracy": 0.8512661065473113,
1098
+ "eval_num_tokens": 2428036.0,
1099
+ "eval_runtime": 86.7869,
1100
+ "eval_samples_per_second": 15.855,
1101
+ "eval_steps_per_second": 1.982,
1102
+ "step": 1040
1103
+ },
1104
+ {
1105
+ "entropy": 0.4824396539479494,
1106
+ "epoch": 2.6376089663760895,
1107
+ "grad_norm": 0.6507360935211182,
1108
+ "learning_rate": 0.00018559389286910275,
1109
+ "loss": 0.4165764808654785,
1110
+ "mean_token_accuracy": 0.8722914069890976,
1111
+ "num_tokens": 2476815.0,
1112
+ "step": 1060
1113
+ },
1114
+ {
1115
+ "epoch": 2.6376089663760895,
1116
+ "eval_entropy": 0.4793398808254752,
1117
+ "eval_loss": 0.5326959490776062,
1118
+ "eval_mean_token_accuracy": 0.8534493650807891,
1119
+ "eval_num_tokens": 2476815.0,
1120
+ "eval_runtime": 86.9559,
1121
+ "eval_samples_per_second": 15.824,
1122
+ "eval_steps_per_second": 1.978,
1123
+ "step": 1060
1124
+ },
1125
+ {
1126
+ "entropy": 0.4605010639876127,
1127
+ "epoch": 2.6874221668742218,
1128
+ "grad_norm": 0.6740535497665405,
1129
+ "learning_rate": 0.00018463600382686253,
1130
+ "loss": 0.4123940944671631,
1131
+ "mean_token_accuracy": 0.8733638986945153,
1132
+ "num_tokens": 2527131.0,
1133
+ "step": 1080
1134
+ },
1135
+ {
1136
+ "epoch": 2.6874221668742218,
1137
+ "eval_entropy": 0.47902208583992584,
1138
+ "eval_loss": 0.5372340083122253,
1139
+ "eval_mean_token_accuracy": 0.851325950303743,
1140
+ "eval_num_tokens": 2527131.0,
1141
+ "eval_runtime": 86.9638,
1142
+ "eval_samples_per_second": 15.823,
1143
+ "eval_steps_per_second": 1.978,
1144
+ "step": 1080
1145
+ },
1146
+ {
1147
+ "entropy": 0.4872019402682781,
1148
+ "epoch": 2.7372353673723535,
1149
+ "grad_norm": 0.6994742155075073,
1150
+ "learning_rate": 0.0001836528239216632,
1151
+ "loss": 0.41599602699279786,
1152
+ "mean_token_accuracy": 0.872775862365961,
1153
+ "num_tokens": 2572537.0,
1154
+ "step": 1100
1155
+ },
1156
+ {
1157
+ "epoch": 2.7372353673723535,
1158
+ "eval_entropy": 0.4893243626453156,
1159
+ "eval_loss": 0.5327795743942261,
1160
+ "eval_mean_token_accuracy": 0.8537560302850812,
1161
+ "eval_num_tokens": 2572537.0,
1162
+ "eval_runtime": 86.823,
1163
+ "eval_samples_per_second": 15.848,
1164
+ "eval_steps_per_second": 1.981,
1165
+ "step": 1100
1166
+ },
1167
+ {
1168
+ "entropy": 0.4949610233306885,
1169
+ "epoch": 2.7870485678704857,
1170
+ "grad_norm": 0.9605912566184998,
1171
+ "learning_rate": 0.0001826446496671543,
1172
+ "loss": 0.4266993045806885,
1173
+ "mean_token_accuracy": 0.8688005246222019,
1174
+ "num_tokens": 2616047.0,
1175
+ "step": 1120
1176
+ },
1177
+ {
1178
+ "epoch": 2.7870485678704857,
1179
+ "eval_entropy": 0.5061517927882283,
1180
+ "eval_loss": 0.5356810092926025,
1181
+ "eval_mean_token_accuracy": 0.8524593568818514,
1182
+ "eval_num_tokens": 2616047.0,
1183
+ "eval_runtime": 86.8385,
1184
+ "eval_samples_per_second": 15.846,
1185
+ "eval_steps_per_second": 1.981,
1186
+ "step": 1120
1187
+ },
1188
+ {
1189
+ "entropy": 0.48401356525719164,
1190
+ "epoch": 2.8368617683686175,
1191
+ "grad_norm": 0.6332499980926514,
1192
+ "learning_rate": 0.00018161178511494022,
1193
+ "loss": 0.42131738662719725,
1194
+ "mean_token_accuracy": 0.8729447312653065,
1195
+ "num_tokens": 2664744.0,
1196
+ "step": 1140
1197
+ },
1198
+ {
1199
+ "epoch": 2.8368617683686175,
1200
+ "eval_entropy": 0.48792897060860035,
1201
+ "eval_loss": 0.5290402173995972,
1202
+ "eval_mean_token_accuracy": 0.853078076659247,
1203
+ "eval_num_tokens": 2664744.0,
1204
+ "eval_runtime": 86.7746,
1205
+ "eval_samples_per_second": 15.857,
1206
+ "eval_steps_per_second": 1.982,
1207
+ "step": 1140
1208
+ },
1209
+ {
1210
+ "entropy": 0.48017631396651267,
1211
+ "epoch": 2.8866749688667497,
1212
+ "grad_norm": 0.8013222217559814,
1213
+ "learning_rate": 0.00018055454176288234,
1214
+ "loss": 0.4195223808288574,
1215
+ "mean_token_accuracy": 0.8721428856253624,
1216
+ "num_tokens": 2710196.0,
1217
+ "step": 1160
1218
+ },
1219
+ {
1220
+ "epoch": 2.8866749688667497,
1221
+ "eval_entropy": 0.5205728571082271,
1222
+ "eval_loss": 0.5242091417312622,
1223
+ "eval_mean_token_accuracy": 0.8535184077052183,
1224
+ "eval_num_tokens": 2710196.0,
1225
+ "eval_runtime": 87.009,
1226
+ "eval_samples_per_second": 15.814,
1227
+ "eval_steps_per_second": 1.977,
1228
+ "step": 1160
1229
+ }
1230
+ ],
1231
+ "logging_steps": 20,
1232
+ "max_steps": 4020,
1233
+ "num_input_tokens_seen": 0,
1234
+ "num_train_epochs": 10,
1235
+ "save_steps": 20,
1236
+ "stateful_callbacks": {
1237
+ "TrainerControl": {
1238
+ "args": {
1239
+ "should_epoch_stop": false,
1240
+ "should_evaluate": false,
1241
+ "should_log": false,
1242
+ "should_save": true,
1243
+ "should_training_stop": false
1244
+ },
1245
+ "attributes": {}
1246
+ }
1247
+ },
1248
+ "total_flos": 1.1472790102826803e+17,
1249
+ "train_batch_size": 4,
1250
+ "trial_name": null,
1251
+ "trial_params": null
1252
+ }