File size: 8,195 Bytes
433c0e5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
{
  "best_global_step": 366,
  "best_metric": 0.41911664605140686,
  "best_model_checkpoint": "training_output/run_20260713_203733/model/checkpoint-366",
  "epoch": 1.0,
  "eval_steps": 500,
  "global_step": 366,
  "is_hyper_param_search": false,
  "is_local_process_zero": true,
  "is_world_process_zero": true,
  "log_history": [
    {
      "epoch": 0.0273224043715847,
      "grad_norm": 4.927176475524902,
      "learning_rate": 7.377049180327868e-07,
      "loss": 0.6899909496307373,
      "step": 10
    },
    {
      "epoch": 0.0546448087431694,
      "grad_norm": 6.0230512619018555,
      "learning_rate": 1.557377049180328e-06,
      "loss": 0.7346426486968994,
      "step": 20
    },
    {
      "epoch": 0.08196721311475409,
      "grad_norm": 6.2535271644592285,
      "learning_rate": 2.377049180327869e-06,
      "loss": 0.6693419456481934,
      "step": 30
    },
    {
      "epoch": 0.1092896174863388,
      "grad_norm": 10.038297653198242,
      "learning_rate": 3.1967213114754097e-06,
      "loss": 0.6888828277587891,
      "step": 40
    },
    {
      "epoch": 0.1366120218579235,
      "grad_norm": 10.041961669921875,
      "learning_rate": 4.016393442622951e-06,
      "loss": 0.623580026626587,
      "step": 50
    },
    {
      "epoch": 0.16393442622950818,
      "grad_norm": 9.107032775878906,
      "learning_rate": 4.836065573770492e-06,
      "loss": 0.692430830001831,
      "step": 60
    },
    {
      "epoch": 0.1912568306010929,
      "grad_norm": 9.090417861938477,
      "learning_rate": 5.655737704918032e-06,
      "loss": 0.6110644817352295,
      "step": 70
    },
    {
      "epoch": 0.2185792349726776,
      "grad_norm": 7.914913177490234,
      "learning_rate": 6.4754098360655735e-06,
      "loss": 0.6268288612365722,
      "step": 80
    },
    {
      "epoch": 0.2459016393442623,
      "grad_norm": 8.18041706085205,
      "learning_rate": 7.2950819672131145e-06,
      "loss": 0.6859075546264648,
      "step": 90
    },
    {
      "epoch": 0.273224043715847,
      "grad_norm": 5.7157392501831055,
      "learning_rate": 8.114754098360657e-06,
      "loss": 0.7717578887939454,
      "step": 100
    },
    {
      "epoch": 0.3005464480874317,
      "grad_norm": 11.109936714172363,
      "learning_rate": 8.934426229508197e-06,
      "loss": 0.6476659774780273,
      "step": 110
    },
    {
      "epoch": 0.32786885245901637,
      "grad_norm": 10.369423866271973,
      "learning_rate": 9.754098360655738e-06,
      "loss": 0.6707509994506836,
      "step": 120
    },
    {
      "epoch": 0.3551912568306011,
      "grad_norm": 10.730874061584473,
      "learning_rate": 1.0573770491803279e-05,
      "loss": 0.6787842750549317,
      "step": 130
    },
    {
      "epoch": 0.3825136612021858,
      "grad_norm": 8.120079040527344,
      "learning_rate": 1.139344262295082e-05,
      "loss": 0.6765446186065673,
      "step": 140
    },
    {
      "epoch": 0.4098360655737705,
      "grad_norm": 14.429889678955078,
      "learning_rate": 1.221311475409836e-05,
      "loss": 0.623370361328125,
      "step": 150
    },
    {
      "epoch": 0.4371584699453552,
      "grad_norm": 9.193517684936523,
      "learning_rate": 1.3032786885245902e-05,
      "loss": 0.6362186908721924,
      "step": 160
    },
    {
      "epoch": 0.4644808743169399,
      "grad_norm": 8.7772855758667,
      "learning_rate": 1.3852459016393443e-05,
      "loss": 0.6641538619995118,
      "step": 170
    },
    {
      "epoch": 0.4918032786885246,
      "grad_norm": 5.2151780128479,
      "learning_rate": 1.4672131147540984e-05,
      "loss": 0.6653701782226562,
      "step": 180
    },
    {
      "epoch": 0.5191256830601093,
      "grad_norm": 17.197834014892578,
      "learning_rate": 1.5491803278688525e-05,
      "loss": 0.6582591533660889,
      "step": 190
    },
    {
      "epoch": 0.546448087431694,
      "grad_norm": 7.26473331451416,
      "learning_rate": 1.6311475409836068e-05,
      "loss": 0.765175724029541,
      "step": 200
    },
    {
      "epoch": 0.5737704918032787,
      "grad_norm": 5.47290563583374,
      "learning_rate": 1.7131147540983607e-05,
      "loss": 0.676247501373291,
      "step": 210
    },
    {
      "epoch": 0.6010928961748634,
      "grad_norm": 4.55684232711792,
      "learning_rate": 1.7950819672131146e-05,
      "loss": 0.6748649597167968,
      "step": 220
    },
    {
      "epoch": 0.6284153005464481,
      "grad_norm": 9.52070140838623,
      "learning_rate": 1.877049180327869e-05,
      "loss": 0.7023877143859864,
      "step": 230
    },
    {
      "epoch": 0.6557377049180327,
      "grad_norm": 10.813623428344727,
      "learning_rate": 1.959016393442623e-05,
      "loss": 0.5075790405273437,
      "step": 240
    },
    {
      "epoch": 0.6830601092896175,
      "grad_norm": 15.77322769165039,
      "learning_rate": 2.040983606557377e-05,
      "loss": 0.5947454929351806,
      "step": 250
    },
    {
      "epoch": 0.7103825136612022,
      "grad_norm": 14.141736030578613,
      "learning_rate": 2.122950819672131e-05,
      "loss": 0.7629669666290283,
      "step": 260
    },
    {
      "epoch": 0.7377049180327869,
      "grad_norm": 0.5761239528656006,
      "learning_rate": 2.2049180327868853e-05,
      "loss": 0.717600679397583,
      "step": 270
    },
    {
      "epoch": 0.7650273224043715,
      "grad_norm": 13.026739120483398,
      "learning_rate": 2.2868852459016393e-05,
      "loss": 1.5919431686401366,
      "step": 280
    },
    {
      "epoch": 0.7923497267759563,
      "grad_norm": 12.15133285522461,
      "learning_rate": 2.3688524590163936e-05,
      "loss": 0.6558570861816406,
      "step": 290
    },
    {
      "epoch": 0.819672131147541,
      "grad_norm": 20.35292625427246,
      "learning_rate": 2.4508196721311478e-05,
      "loss": 0.6991987705230713,
      "step": 300
    },
    {
      "epoch": 0.8469945355191257,
      "grad_norm": 9.59134292602539,
      "learning_rate": 2.5327868852459018e-05,
      "loss": 1.1239334106445313,
      "step": 310
    },
    {
      "epoch": 0.8743169398907104,
      "grad_norm": 4.265232086181641,
      "learning_rate": 2.6147540983606557e-05,
      "loss": 0.4866987705230713,
      "step": 320
    },
    {
      "epoch": 0.9016393442622951,
      "grad_norm": 9.87460708618164,
      "learning_rate": 2.69672131147541e-05,
      "loss": 0.7435293674468995,
      "step": 330
    },
    {
      "epoch": 0.9289617486338798,
      "grad_norm": 12.707701683044434,
      "learning_rate": 2.7786885245901642e-05,
      "loss": 0.9141542434692382,
      "step": 340
    },
    {
      "epoch": 0.9562841530054644,
      "grad_norm": 2.9138967990875244,
      "learning_rate": 2.860655737704918e-05,
      "loss": 0.5319347858428956,
      "step": 350
    },
    {
      "epoch": 0.9836065573770492,
      "grad_norm": 6.917504787445068,
      "learning_rate": 2.942622950819672e-05,
      "loss": 0.5438683509826661,
      "step": 360
    },
    {
      "epoch": 1.0,
      "eval_accuracy": 0.8492,
      "eval_confusion_matrix": [
        [
          282,
          0
        ],
        [
          54,
          22
        ]
      ],
      "eval_f1": 0.449,
      "eval_loss": 0.41911664605140686,
      "eval_pr_auc": 0.6187,
      "eval_precision": 1.0,
      "eval_recall": 0.2895,
      "eval_roc_auc": 0.7485,
      "eval_runtime": 138.7598,
      "eval_samples_per_second": 2.58,
      "eval_steps_per_second": 0.649,
      "step": 366
    }
  ],
  "logging_steps": 10,
  "max_steps": 3660,
  "num_input_tokens_seen": 0,
  "num_train_epochs": 10,
  "save_steps": 500,
  "stateful_callbacks": {
    "EarlyStoppingCallback": {
      "args": {
        "early_stopping_patience": 3,
        "early_stopping_threshold": 0.0
      },
      "attributes": {
        "early_stopping_patience_counter": 0
      }
    },
    "TrainerControl": {
      "args": {
        "should_epoch_stop": false,
        "should_evaluate": false,
        "should_log": false,
        "should_save": true,
        "should_training_stop": false
      },
      "attributes": {}
    }
  },
  "total_flos": 385194585047040.0,
  "train_batch_size": 4,
  "trial_name": null,
  "trial_params": null
}