bimabk commited on
Commit
4940b17
·
verified ·
1 Parent(s): afe908f

Upload task output 1

Browse files
adapter_config.json CHANGED
@@ -29,13 +29,13 @@
29
  "rank_pattern": {},
30
  "revision": null,
31
  "target_modules": [
 
32
  "gate_proj",
33
- "k_proj",
34
- "v_proj",
35
  "o_proj",
36
- "q_proj",
 
37
  "up_proj",
38
- "down_proj"
39
  ],
40
  "target_parameters": null,
41
  "task_type": "CAUSAL_LM",
 
29
  "rank_pattern": {},
30
  "revision": null,
31
  "target_modules": [
32
+ "q_proj",
33
  "gate_proj",
 
 
34
  "o_proj",
35
+ "down_proj",
36
+ "k_proj",
37
  "up_proj",
38
+ "v_proj"
39
  ],
40
  "target_parameters": null,
41
  "task_type": "CAUSAL_LM",
adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:377dc67b8d9ac5d2e476c15f9cf0609b27690828cfdd9f4491dba511d66d3ae0
3
  size 957942768
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c9bc32fac7f25bcede80af9be64db4efa53407d9dbd6b511856089e76298e8f7
3
  size 957942768
loss.txt CHANGED
@@ -1 +1 @@
1
- 68,no_eval
 
1
+ 75,-0.5700000166893006
trainer_state.json CHANGED
@@ -2,446 +2,545 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.0102010201020102,
6
  "eval_steps": 500,
7
- "global_step": 68,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "clip_ratio/high_max": 0.0,
14
- "clip_ratio/high_mean": 0.0,
15
- "clip_ratio/low_mean": 0.0,
16
  "clip_ratio/low_min": 0.0,
17
- "clip_ratio/region_mean": 0.0,
18
  "completions/clipped_ratio": 0.0,
19
- "completions/max_length": 374.4,
20
- "completions/max_terminated_length": 374.4,
21
- "completions/mean_length": 291.50001220703126,
22
- "completions/mean_terminated_length": 291.50001220703126,
23
- "completions/min_length": 174.8,
24
- "completions/min_terminated_length": 174.8,
25
- "entropy": 0.737431001663208,
26
- "epoch": 0.00075007500750075,
27
- "frac_reward_zero_std": 0.6533333778381347,
28
- "grad_norm": 0.369140625,
29
- "kl": 0.00885026976466179,
30
  "learning_rate": 1.137216e-06,
31
- "loss": 0.00034544954542070627,
32
- "num_tokens": 126703.0,
33
- "reward": 0.0383333370089531,
34
- "reward_std": 0.05835988484323025,
35
- "rewards/env_goofspiel_reward/mean": 0.03833333514630795,
36
- "rewards/env_goofspiel_reward/std": 0.15159520953893663,
37
- "sampling/importance_sampling_ratio/max": 1.7607571125030517,
38
- "sampling/importance_sampling_ratio/mean": 0.9916059970855713,
39
- "sampling/importance_sampling_ratio/min": 0.5116463124752044,
40
- "sampling/sampling_logp_difference/max": 0.6887424349784851,
41
- "sampling/sampling_logp_difference/mean": 0.06558751240372658,
42
  "step": 5,
43
- "step_time": 3.541577606199735
44
  },
45
  {
46
- "clip_ratio/high_max": 0.0,
47
- "clip_ratio/high_mean": 0.0,
48
- "clip_ratio/low_mean": 0.0,
49
  "clip_ratio/low_min": 0.0,
50
- "clip_ratio/region_mean": 0.0,
51
  "completions/clipped_ratio": 0.0,
52
- "completions/max_length": 373.6,
53
- "completions/max_terminated_length": 373.6,
54
- "completions/mean_length": 292.0866760253906,
55
- "completions/mean_terminated_length": 292.0866760253906,
56
- "completions/min_length": 179.8,
57
- "completions/min_terminated_length": 179.8,
58
- "entropy": 0.7154561956723531,
59
- "epoch": 0.0015001500150015,
60
- "frac_reward_zero_std": 0.6666666865348816,
61
- "grad_norm": 0.462890625,
62
- "kl": 0.009678552253171801,
63
  "learning_rate": 2.5587359999999995e-06,
64
- "loss": -3.925009514205158e-05,
65
- "num_tokens": 254525.0,
66
- "reward": 0.03886667159385979,
67
- "reward_std": 0.05817132312804461,
68
- "rewards/env_goofspiel_reward/mean": 0.03886666861362755,
69
- "rewards/env_goofspiel_reward/std": 0.14201362472958862,
70
- "sampling/importance_sampling_ratio/max": 1.9776098012924195,
71
- "sampling/importance_sampling_ratio/mean": 1.0280600190162659,
72
- "sampling/importance_sampling_ratio/min": 0.4622887670993805,
73
- "sampling/sampling_logp_difference/max": 0.8463176250457763,
74
- "sampling/sampling_logp_difference/mean": 0.06613281443715095,
75
  "step": 10,
76
- "step_time": 3.1668822380001074
77
  },
78
  {
79
- "clip_ratio/high_max": 0.0,
80
- "clip_ratio/high_mean": 0.0,
81
- "clip_ratio/low_mean": 0.0,
82
  "clip_ratio/low_min": 0.0,
83
- "clip_ratio/region_mean": 0.0,
84
  "completions/clipped_ratio": 0.0,
85
- "completions/max_length": 373.6,
86
- "completions/max_terminated_length": 373.6,
87
- "completions/mean_length": 288.72001953125,
88
- "completions/mean_terminated_length": 288.72001953125,
89
- "completions/min_length": 187.2,
90
- "completions/min_terminated_length": 187.2,
91
- "entropy": 0.6402347763379415,
92
- "epoch": 0.0022502250225022503,
93
- "frac_reward_zero_std": 0.8000000476837158,
94
- "grad_norm": 0.412109375,
95
- "kl": 0.00951577623685201,
96
  "learning_rate": 3.9802559999999995e-06,
97
- "loss": 0.00047838701866567137,
98
- "num_tokens": 380635.0,
99
- "reward": 0.07146667279303073,
100
- "reward_std": 0.09088679552078247,
101
- "rewards/env_goofspiel_reward/mean": 0.07146666683256626,
102
- "rewards/env_goofspiel_reward/std": 0.22424434274435043,
103
- "sampling/importance_sampling_ratio/max": 1.5242333889007569,
104
- "sampling/importance_sampling_ratio/mean": 0.9860278725624084,
105
- "sampling/importance_sampling_ratio/min": 0.5668014168739319,
106
- "sampling/sampling_logp_difference/max": 0.5752804994583129,
107
- "sampling/sampling_logp_difference/mean": 0.05473804771900177,
108
  "step": 15,
109
- "step_time": 3.1726541418000123
110
  },
111
  {
112
- "clip_ratio/high_max": 0.0,
113
- "clip_ratio/high_mean": 0.0,
114
- "clip_ratio/low_mean": 0.0,
115
  "clip_ratio/low_min": 0.0,
116
- "clip_ratio/region_mean": 0.0,
117
  "completions/clipped_ratio": 0.0,
118
- "completions/max_length": 373.2,
119
- "completions/max_terminated_length": 373.2,
120
- "completions/mean_length": 273.82001037597655,
121
- "completions/mean_terminated_length": 273.82001037597655,
122
- "completions/min_length": 200.2,
123
- "completions/min_terminated_length": 200.2,
124
- "entropy": 0.5869409064451854,
125
- "epoch": 0.003000300030003,
126
- "frac_reward_zero_std": 0.7600000500679016,
127
- "grad_norm": 0.158203125,
128
- "kl": 0.029618356159577766,
129
  "learning_rate": 5.401775999999999e-06,
130
- "loss": 0.00020813762675970794,
131
- "num_tokens": 501743.0,
132
- "reward": 0.06360000669956208,
133
- "reward_std": 0.07976165302097797,
134
- "rewards/env_goofspiel_reward/mean": 0.06360000558197498,
135
- "rewards/env_goofspiel_reward/std": 0.20161283165216445,
136
- "sampling/importance_sampling_ratio/max": 1.7165520668029786,
137
- "sampling/importance_sampling_ratio/mean": 0.9985299944877625,
138
- "sampling/importance_sampling_ratio/min": 0.6464852690696716,
139
- "sampling/sampling_logp_difference/max": 0.6010382175445557,
140
- "sampling/sampling_logp_difference/mean": 0.05117494091391563,
141
  "step": 20,
142
- "step_time": 3.1184157538001274
143
  },
144
  {
145
- "clip_ratio/high_max": 0.0,
146
- "clip_ratio/high_mean": 0.0,
147
- "clip_ratio/low_mean": 0.0,
148
- "clip_ratio/low_min": 0.0,
149
- "clip_ratio/region_mean": 0.0,
150
  "completions/clipped_ratio": 0.0,
151
- "completions/max_length": 374.2,
152
- "completions/max_terminated_length": 374.2,
153
- "completions/mean_length": 291.16668090820315,
154
- "completions/mean_terminated_length": 291.16668090820315,
155
- "completions/min_length": 212.0,
156
- "completions/min_terminated_length": 212.0,
157
- "entropy": 0.5825250327587128,
158
- "epoch": 0.0037503750375037503,
159
- "frac_reward_zero_std": 0.8133333921432495,
160
- "grad_norm": 0.30859375,
161
- "kl": 0.03789825042088826,
162
  "learning_rate": 6.8232959999999994e-06,
163
- "loss": -5.795806646347046e-05,
164
- "num_tokens": 629207.0,
165
- "reward": 0.05593333579599857,
166
- "reward_std": 0.0791016798466444,
167
- "rewards/env_goofspiel_reward/mean": 0.055933335050940516,
168
- "rewards/env_goofspiel_reward/std": 0.16461323350667953,
169
- "sampling/importance_sampling_ratio/max": 1.6597614526748656,
170
- "sampling/importance_sampling_ratio/mean": 0.9899526715278626,
171
- "sampling/importance_sampling_ratio/min": 0.5461978197097779,
172
- "sampling/sampling_logp_difference/max": 0.5474630713462829,
173
- "sampling/sampling_logp_difference/mean": 0.056329603493213656,
174
  "step": 25,
175
- "step_time": 3.1455567411999255
176
  },
177
  {
178
- "clip_ratio/high_max": 0.0,
179
- "clip_ratio/high_mean": 0.0,
180
- "clip_ratio/low_mean": 0.0,
181
- "clip_ratio/low_min": 0.0,
182
- "clip_ratio/region_mean": 0.0,
183
  "completions/clipped_ratio": 0.0,
184
- "completions/max_length": 373.8,
185
- "completions/max_terminated_length": 373.8,
186
- "completions/mean_length": 288.2733459472656,
187
- "completions/mean_terminated_length": 288.2733459472656,
188
- "completions/min_length": 205.4,
189
- "completions/min_terminated_length": 205.4,
190
- "entropy": 0.6367802540461223,
191
- "epoch": 0.004500450045004501,
192
- "frac_reward_zero_std": 0.8133333921432495,
193
- "grad_norm": 0.3203125,
194
- "kl": 0.15099084513882796,
195
  "learning_rate": 8.244816e-06,
196
- "loss": 2.9896479099988936e-05,
197
- "num_tokens": 755141.0,
198
- "reward": 0.04773333668708801,
199
- "reward_std": 0.056945668533444405,
200
- "rewards/env_goofspiel_reward/mean": 0.04773333407938481,
201
- "rewards/env_goofspiel_reward/std": 0.1535952940583229,
202
- "sampling/importance_sampling_ratio/max": 1.4472879409790038,
203
- "sampling/importance_sampling_ratio/mean": 1.000359261035919,
204
- "sampling/importance_sampling_ratio/min": 0.6299581527709961,
205
- "sampling/sampling_logp_difference/max": 0.49155421257019044,
206
- "sampling/sampling_logp_difference/mean": 0.04638729840517044,
207
  "step": 30,
208
- "step_time": 3.104708982399825
209
  },
210
  {
211
- "clip_ratio/high_max": 0.0,
212
- "clip_ratio/high_mean": 0.0,
213
- "clip_ratio/low_mean": 0.0,
214
- "clip_ratio/low_min": 0.0,
215
- "clip_ratio/region_mean": 0.0,
216
  "completions/clipped_ratio": 0.0,
217
- "completions/max_length": 373.4,
218
- "completions/max_terminated_length": 373.4,
219
- "completions/mean_length": 294.12001953125,
220
- "completions/mean_terminated_length": 294.12001953125,
221
  "completions/min_length": 212.0,
222
  "completions/min_terminated_length": 212.0,
223
- "entropy": 0.6307838877042135,
224
- "epoch": 0.005250525052505251,
225
- "frac_reward_zero_std": 0.8133333921432495,
226
- "grad_norm": 0.2041015625,
227
- "kl": 0.21180881708860397,
228
  "learning_rate": 9.666336e-06,
229
- "loss": 0.0003116762964054942,
230
- "num_tokens": 881937.0,
231
- "reward": 0.04366667197318748,
232
- "reward_std": 0.0626968042459339,
233
- "rewards/env_goofspiel_reward/mean": 0.04366666899295524,
234
- "rewards/env_goofspiel_reward/std": 0.14476882207673042,
235
- "sampling/importance_sampling_ratio/max": 1.540164875984192,
236
- "sampling/importance_sampling_ratio/mean": 1.007632350921631,
237
- "sampling/importance_sampling_ratio/min": 0.6116935849189759,
238
- "sampling/sampling_logp_difference/max": 0.5826021313667298,
239
- "sampling/sampling_logp_difference/mean": 0.054141230136156085,
240
  "step": 35,
241
- "step_time": 3.152498085000025
242
  },
243
  {
244
- "clip_ratio/high_max": 0.0,
245
- "clip_ratio/high_mean": 0.0,
246
- "clip_ratio/low_mean": 0.0,
247
  "clip_ratio/low_min": 0.0,
248
- "clip_ratio/region_mean": 0.0,
249
  "completions/clipped_ratio": 0.0,
250
- "completions/max_length": 365.8,
251
- "completions/max_terminated_length": 365.8,
252
- "completions/mean_length": 281.62001953125,
253
- "completions/mean_terminated_length": 281.62001953125,
254
- "completions/min_length": 212.0,
255
- "completions/min_terminated_length": 212.0,
256
- "entropy": 0.553737320502599,
257
- "epoch": 0.006000600060006,
258
- "frac_reward_zero_std": 0.8266667246818542,
259
- "grad_norm": 0.0888671875,
260
- "kl": 0.5656270523866017,
261
- "learning_rate": 9.950639260700543e-06,
262
- "loss": 0.00017667778301984073,
263
- "num_tokens": 1005294.0,
264
- "reward": 0.056000004336237905,
265
- "reward_std": 0.07919596247375012,
266
- "rewards/env_goofspiel_reward/mean": 0.05600000284612179,
267
- "rewards/env_goofspiel_reward/std": 0.18038542717695236,
268
- "sampling/importance_sampling_ratio/max": 1.5142346620559692,
269
- "sampling/importance_sampling_ratio/mean": 0.9889694690704346,
270
- "sampling/importance_sampling_ratio/min": 0.6695701956748963,
271
- "sampling/sampling_logp_difference/max": 0.45929052233695983,
272
- "sampling/sampling_logp_difference/mean": 0.05039779171347618,
273
  "step": 40,
274
- "step_time": 3.061584199799836
275
  },
276
  {
277
- "clip_ratio/high_max": 0.0,
278
- "clip_ratio/high_mean": 0.0,
279
- "clip_ratio/low_mean": 0.0,
280
- "clip_ratio/low_min": 0.0,
281
- "clip_ratio/region_mean": 0.0,
282
  "completions/clipped_ratio": 0.0,
283
- "completions/max_length": 374.0,
284
- "completions/max_terminated_length": 374.0,
285
- "completions/mean_length": 303.58001708984375,
286
- "completions/mean_terminated_length": 303.58001708984375,
287
- "completions/min_length": 212.0,
288
- "completions/min_terminated_length": 212.0,
289
- "entropy": 0.48466593225797017,
290
- "epoch": 0.0067506750675067504,
291
- "frac_reward_zero_std": 0.8266667246818542,
292
- "grad_norm": 0.1748046875,
293
- "kl": 1.0889011422793071,
294
- "learning_rate": 9.950636257297004e-06,
295
- "loss": 0.00040990984998643396,
296
- "num_tokens": 1135969.0,
297
- "reward": 0.04786667115986347,
298
- "reward_std": 0.06807081587612629,
299
- "rewards/env_goofspiel_reward/mean": 0.047866668179631235,
300
- "rewards/env_goofspiel_reward/std": 0.1721431568264961,
301
- "sampling/importance_sampling_ratio/max": 1.4374094486236573,
302
- "sampling/importance_sampling_ratio/mean": 0.9694242954254151,
303
- "sampling/importance_sampling_ratio/min": 0.6029698491096497,
304
- "sampling/sampling_logp_difference/max": 0.5160395622253418,
305
- "sampling/sampling_logp_difference/mean": 0.04300674088299274,
306
  "step": 45,
307
- "step_time": 3.19344236120005
308
  },
309
  {
310
- "clip_ratio/high_max": 0.0,
311
- "clip_ratio/high_mean": 0.0,
312
- "clip_ratio/low_mean": 0.0,
313
- "clip_ratio/low_min": 0.0,
314
- "clip_ratio/region_mean": 0.0,
315
  "completions/clipped_ratio": 0.0,
316
- "completions/max_length": 375.6,
317
- "completions/max_terminated_length": 375.6,
318
- "completions/mean_length": 296.2133483886719,
319
- "completions/mean_terminated_length": 296.2133483886719,
320
- "completions/min_length": 212.0,
321
- "completions/min_terminated_length": 212.0,
322
- "entropy": 0.3623567740122477,
323
- "epoch": 0.007500750075007501,
324
- "frac_reward_zero_std": 0.8800000429153443,
325
- "grad_norm": 0.134765625,
326
- "kl": 0.9536839803059896,
327
- "learning_rate": 9.950630943585028e-06,
328
- "loss": 0.00046116397716104983,
329
- "num_tokens": 1262891.0,
330
- "reward": 0.040000003576278684,
331
- "reward_std": 0.05656854659318924,
332
- "rewards/env_goofspiel_reward/mean": 0.04000000208616257,
333
- "rewards/env_goofspiel_reward/std": 0.13237088024616242,
334
- "sampling/importance_sampling_ratio/max": 1.4614337921142577,
335
- "sampling/importance_sampling_ratio/mean": 0.9917921781539917,
336
- "sampling/importance_sampling_ratio/min": 0.7945461988449096,
337
- "sampling/sampling_logp_difference/max": 0.34459147453308103,
338
- "sampling/sampling_logp_difference/mean": 0.023883017525076867,
339
  "step": 50,
340
- "step_time": 3.080125956599841
341
  },
342
  {
343
- "clip_ratio/high_max": 0.0,
344
- "clip_ratio/high_mean": 0.0,
345
- "clip_ratio/low_mean": 0.0,
346
  "clip_ratio/low_min": 0.0,
347
- "clip_ratio/region_mean": 0.0,
348
  "completions/clipped_ratio": 0.0,
349
- "completions/max_length": 373.8,
350
- "completions/max_terminated_length": 373.8,
351
- "completions/mean_length": 297.3133483886719,
352
- "completions/mean_terminated_length": 297.3133483886719,
353
- "completions/min_length": 212.0,
354
- "completions/min_terminated_length": 212.0,
355
- "entropy": 0.412084698677063,
356
- "epoch": 0.00825082508250825,
357
- "frac_reward_zero_std": 0.840000057220459,
358
- "grad_norm": 0.11279296875,
359
- "kl": 0.8989204486211141,
360
- "learning_rate": 9.950623319567901e-06,
361
- "loss": 0.00017252418911084533,
362
- "num_tokens": 1391274.0,
363
- "reward": 0.059933338314294815,
364
- "reward_std": 0.07363338991999627,
365
- "rewards/env_goofspiel_reward/mean": 0.059933335334062574,
366
- "rewards/env_goofspiel_reward/std": 0.19352754354476928,
367
- "sampling/importance_sampling_ratio/max": 1.510833191871643,
368
- "sampling/importance_sampling_ratio/mean": 1.0025275230407715,
369
- "sampling/importance_sampling_ratio/min": 0.7753824949264526,
370
- "sampling/sampling_logp_difference/max": 0.37447161674499513,
371
- "sampling/sampling_logp_difference/mean": 0.025511124357581138,
372
  "step": 55,
373
- "step_time": 3.1612697836000736
374
  },
375
  {
376
- "clip_ratio/high_max": 0.0,
377
- "clip_ratio/high_mean": 0.0,
378
- "clip_ratio/low_mean": 0.0,
379
  "clip_ratio/low_min": 0.0,
380
- "clip_ratio/region_mean": 0.0,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
381
  "completions/clipped_ratio": 0.0,
382
  "completions/max_length": 374.0,
383
  "completions/max_terminated_length": 374.0,
384
- "completions/mean_length": 293.233349609375,
385
- "completions/mean_terminated_length": 293.233349609375,
386
- "completions/min_length": 212.0,
387
- "completions/min_terminated_length": 212.0,
388
- "entropy": 0.3592885712782542,
389
- "epoch": 0.009000900090009001,
390
- "frac_reward_zero_std": 0.8533333778381348,
391
- "grad_norm": 0.0164794921875,
392
- "kl": 0.9290923396746318,
393
- "learning_rate": 9.950613385250344e-06,
394
- "loss": 0.00017089147586375476,
395
- "num_tokens": 1517875.0,
396
- "reward": 0.06400000602006913,
397
- "reward_std": 0.07919596470892429,
398
- "rewards/env_goofspiel_reward/mean": 0.06400000229477883,
399
- "rewards/env_goofspiel_reward/std": 0.192934051156044,
400
- "sampling/importance_sampling_ratio/max": 1.3080760478973388,
401
- "sampling/importance_sampling_ratio/mean": 1.00157231092453,
402
- "sampling/importance_sampling_ratio/min": 0.809451448917389,
403
- "sampling/sampling_logp_difference/max": 0.3210816144943237,
404
- "sampling/sampling_logp_difference/mean": 0.021189583651721477,
405
- "step": 60,
406
- "step_time": 3.1453409107998596
407
  },
408
  {
409
- "clip_ratio/high_max": 0.0,
410
- "clip_ratio/high_mean": 0.0,
411
- "clip_ratio/low_mean": 0.0,
412
  "clip_ratio/low_min": 0.0,
413
- "clip_ratio/region_mean": 0.0,
414
  "completions/clipped_ratio": 0.0,
415
  "completions/max_length": 374.0,
416
  "completions/max_terminated_length": 374.0,
417
- "completions/mean_length": 280.706689453125,
418
- "completions/mean_terminated_length": 280.706689453125,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
419
  "completions/min_length": 212.0,
420
  "completions/min_terminated_length": 212.0,
421
- "entropy": 0.46913262009620665,
422
- "epoch": 0.00975097509750975,
423
- "frac_reward_zero_std": 0.840000057220459,
424
- "grad_norm": 0.04541015625,
425
- "kl": 1.9159570078055064,
426
- "learning_rate": 9.95060114063851e-06,
427
- "loss": 0.00043630152940750123,
428
- "num_tokens": 1640903.0,
429
- "reward": 0.07593334019184113,
430
- "reward_std": 0.07363339141011238,
431
- "rewards/env_goofspiel_reward/mean": 0.07593333572149277,
432
- "rewards/env_goofspiel_reward/std": 0.22238647043704987,
433
- "sampling/importance_sampling_ratio/max": 1.42401442527771,
434
- "sampling/importance_sampling_ratio/mean": 1.001141333580017,
435
- "sampling/importance_sampling_ratio/min": 0.7154132127761841,
436
- "sampling/sampling_logp_difference/max": 0.3389713287353516,
437
- "sampling/sampling_logp_difference/mean": 0.026769372820854186,
438
- "step": 65,
439
- "step_time": 3.0763218426000094
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
440
  }
441
  ],
442
  "logging_steps": 5,
443
- "max_steps": 19998,
444
- "num_input_tokens_seen": 1713916,
445
  "num_train_epochs": 3,
446
  "save_steps": 500,
447
  "stateful_callbacks": {
@@ -451,13 +550,13 @@
451
  "should_evaluate": false,
452
  "should_log": false,
453
  "should_save": true,
454
- "should_training_stop": true
455
  },
456
  "attributes": {}
457
  }
458
  },
459
  "total_flos": 0.0,
460
- "train_batch_size": 10,
461
  "trial_name": null,
462
  "trial_params": null
463
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.012,
6
  "eval_steps": 500,
7
+ "global_step": 75,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "clip_ratio/high_max": 0.014437134563922881,
14
+ "clip_ratio/high_mean": 0.007218567281961441,
15
+ "clip_ratio/low_mean": 0.005763888917863369,
16
  "clip_ratio/low_min": 0.0,
17
+ "clip_ratio/region_mean": 0.01298245619982481,
18
  "completions/clipped_ratio": 0.0,
19
+ "completions/max_length": 374.0,
20
+ "completions/max_terminated_length": 374.0,
21
+ "completions/mean_length": 294.78125,
22
+ "completions/mean_terminated_length": 294.78125,
23
+ "completions/min_length": 189.6,
24
+ "completions/min_terminated_length": 189.6,
25
+ "entropy": 0.35714133381843566,
26
+ "epoch": 0.0008,
27
+ "frac_reward_zero_std": 0.475,
28
+ "grad_norm": 0.14499153196811676,
29
+ "kl": 0.006699140788987279,
30
  "learning_rate": 1.137216e-06,
31
+ "loss": 0.0004814713727682829,
32
+ "num_tokens": 136090.0,
33
+ "reward": 0.292125004529953,
34
+ "reward_std": 0.2656953722238541,
35
+ "rewards/env_goofspiel_reward/mean": 0.292125004529953,
36
+ "rewards/env_goofspiel_reward/std": 0.41643730401992796,
37
+ "sampling/importance_sampling_ratio/max": 1.8895051956176758,
38
+ "sampling/importance_sampling_ratio/mean": 0.9195514798164368,
39
+ "sampling/importance_sampling_ratio/min": 0.2350650832056999,
40
+ "sampling/sampling_logp_difference/max": 1.6994480609893798,
41
+ "sampling/sampling_logp_difference/mean": 0.09322866201400756,
42
  "step": 5,
43
+ "step_time": 5.709857220399681
44
  },
45
  {
46
+ "clip_ratio/high_max": 0.02671568635851145,
47
+ "clip_ratio/high_mean": 0.013357843179255724,
48
+ "clip_ratio/low_mean": 0.016053921636193992,
49
  "clip_ratio/low_min": 0.0,
50
+ "clip_ratio/region_mean": 0.029411764815449715,
51
  "completions/clipped_ratio": 0.0,
52
+ "completions/max_length": 374.2,
53
+ "completions/max_terminated_length": 374.2,
54
+ "completions/mean_length": 290.54375,
55
+ "completions/mean_terminated_length": 290.54375,
56
+ "completions/min_length": 194.4,
57
+ "completions/min_terminated_length": 194.4,
58
+ "entropy": 0.36814531981945037,
59
+ "epoch": 0.0016,
60
+ "frac_reward_zero_std": 0.55,
61
+ "grad_norm": 0.15409794449806213,
62
+ "kl": 0.024915735074318945,
63
  "learning_rate": 2.5587359999999995e-06,
64
+ "loss": 0.00025723695289343597,
65
+ "num_tokens": 271136.0,
66
+ "reward": 0.30362500846385954,
67
+ "reward_std": 0.21761211454868318,
68
+ "rewards/env_goofspiel_reward/mean": 0.30362500846385954,
69
+ "rewards/env_goofspiel_reward/std": 0.40538435578346255,
70
+ "sampling/importance_sampling_ratio/max": 2.3090060234069822,
71
+ "sampling/importance_sampling_ratio/mean": 0.9982686519622803,
72
+ "sampling/importance_sampling_ratio/min": 0.11310269832611083,
73
+ "sampling/sampling_logp_difference/max": 1.6586394786834717,
74
+ "sampling/sampling_logp_difference/mean": 0.09253094047307968,
75
  "step": 10,
76
+ "step_time": 5.373984831200687
77
  },
78
  {
79
+ "clip_ratio/high_max": 0.030514705926179886,
80
+ "clip_ratio/high_mean": 0.01672794120386243,
81
+ "clip_ratio/low_mean": 0.01482843142002821,
82
  "clip_ratio/low_min": 0.0,
83
+ "clip_ratio/region_mean": 0.03155637262389064,
84
  "completions/clipped_ratio": 0.0,
85
+ "completions/max_length": 375.6,
86
+ "completions/max_terminated_length": 375.6,
87
+ "completions/mean_length": 283.45625,
88
+ "completions/mean_terminated_length": 283.45625,
89
+ "completions/min_length": 194.6,
90
+ "completions/min_terminated_length": 194.6,
91
+ "entropy": 0.38065551668405534,
92
+ "epoch": 0.0024,
93
+ "frac_reward_zero_std": 0.4375,
94
+ "grad_norm": 0.0815606489777565,
95
+ "kl": 0.02344995441380888,
96
  "learning_rate": 3.9802559999999995e-06,
97
+ "loss": 0.000516003929078579,
98
+ "num_tokens": 403825.0,
99
+ "reward": 0.35987500548362733,
100
+ "reward_std": 0.2653418242931366,
101
+ "rewards/env_goofspiel_reward/mean": 0.35987500548362733,
102
+ "rewards/env_goofspiel_reward/std": 0.4154684245586395,
103
+ "sampling/importance_sampling_ratio/max": 1.841845488548279,
104
+ "sampling/importance_sampling_ratio/mean": 0.9473352670669556,
105
+ "sampling/importance_sampling_ratio/min": 0.2071388103067875,
106
+ "sampling/sampling_logp_difference/max": 1.491676390171051,
107
+ "sampling/sampling_logp_difference/mean": 0.09024534374475479,
108
  "step": 15,
109
+ "step_time": 5.279587671000627
110
  },
111
  {
112
+ "clip_ratio/high_max": 0.023333333432674408,
113
+ "clip_ratio/high_mean": 0.011666666716337204,
114
+ "clip_ratio/low_mean": 0.012774122878909111,
115
  "clip_ratio/low_min": 0.0,
116
+ "clip_ratio/region_mean": 0.024440789688378574,
117
  "completions/clipped_ratio": 0.0,
118
+ "completions/max_length": 373.8,
119
+ "completions/max_terminated_length": 373.8,
120
+ "completions/mean_length": 279.5875,
121
+ "completions/mean_terminated_length": 279.5875,
122
+ "completions/min_length": 206.8,
123
+ "completions/min_terminated_length": 206.8,
124
+ "entropy": 0.3506437622010708,
125
+ "epoch": 0.0032,
126
+ "frac_reward_zero_std": 0.4625,
127
+ "grad_norm": 0.12220246344804764,
128
+ "kl": 0.23682632837444545,
129
  "learning_rate": 5.401775999999999e-06,
130
+ "loss": -0.0002838193904608488,
131
+ "num_tokens": 535747.0,
132
+ "reward": 0.374812513589859,
133
+ "reward_std": 0.24421700537204744,
134
+ "rewards/env_goofspiel_reward/mean": 0.374812513589859,
135
+ "rewards/env_goofspiel_reward/std": 0.39649735689163207,
136
+ "sampling/importance_sampling_ratio/max": 2.3718762159347535,
137
+ "sampling/importance_sampling_ratio/mean": 0.9897154331207275,
138
+ "sampling/importance_sampling_ratio/min": 0.2213693767786026,
139
+ "sampling/sampling_logp_difference/max": 2.0126638174057008,
140
+ "sampling/sampling_logp_difference/mean": 0.10554229319095612,
141
  "step": 20,
142
+ "step_time": 5.332370807199913
143
  },
144
  {
145
+ "clip_ratio/high_max": 0.029443860985338688,
146
+ "clip_ratio/high_mean": 0.016110819298774004,
147
+ "clip_ratio/low_mean": 0.027831450570374727,
148
+ "clip_ratio/low_min": 0.011572128906846047,
149
+ "clip_ratio/region_mean": 0.04394227024167776,
150
  "completions/clipped_ratio": 0.0,
151
+ "completions/max_length": 374.4,
152
+ "completions/max_terminated_length": 374.4,
153
+ "completions/mean_length": 301.0375,
154
+ "completions/mean_terminated_length": 301.0375,
155
+ "completions/min_length": 218.8,
156
+ "completions/min_terminated_length": 218.8,
157
+ "entropy": 0.3545067012310028,
158
+ "epoch": 0.004,
159
+ "frac_reward_zero_std": 0.35,
160
+ "grad_norm": 0.09991537779569626,
161
+ "kl": 0.6746378809213638,
162
  "learning_rate": 6.8232959999999994e-06,
163
+ "loss": -0.0003991848789155483,
164
+ "num_tokens": 673746.0,
165
+ "reward": 0.34875001609325407,
166
+ "reward_std": 0.3128947615623474,
167
+ "rewards/env_goofspiel_reward/mean": 0.34875001609325407,
168
+ "rewards/env_goofspiel_reward/std": 0.3975376784801483,
169
+ "sampling/importance_sampling_ratio/max": 2.3523300170898436,
170
+ "sampling/importance_sampling_ratio/mean": 0.9350853443145752,
171
+ "sampling/importance_sampling_ratio/min": 0.03211224116384983,
172
+ "sampling/sampling_logp_difference/max": 2.597071409225464,
173
+ "sampling/sampling_logp_difference/mean": 0.14846422374248505,
174
  "step": 25,
175
+ "step_time": 5.470841645199835
176
  },
177
  {
178
+ "clip_ratio/high_max": 0.015093954280018806,
179
+ "clip_ratio/high_mean": 0.007546977140009403,
180
+ "clip_ratio/low_mean": 0.01916505442932248,
181
+ "clip_ratio/low_min": 0.005625000037252903,
182
+ "clip_ratio/region_mean": 0.026712031569331884,
183
  "completions/clipped_ratio": 0.0,
184
+ "completions/max_length": 374.0,
185
+ "completions/max_terminated_length": 374.0,
186
+ "completions/mean_length": 283.44375,
187
+ "completions/mean_terminated_length": 283.44375,
188
+ "completions/min_length": 212.0,
189
+ "completions/min_terminated_length": 212.0,
190
+ "entropy": 0.3386394247412682,
191
+ "epoch": 0.0048,
192
+ "frac_reward_zero_std": 0.5625,
193
+ "grad_norm": 0.05208470672369003,
194
+ "kl": 3.1543088920414446,
195
  "learning_rate": 8.244816e-06,
196
+ "loss": 3.84216895326972e-05,
197
+ "num_tokens": 805457.0,
198
+ "reward": 0.41250001192092894,
199
+ "reward_std": 0.2121320277452469,
200
+ "rewards/env_goofspiel_reward/mean": 0.41250001192092894,
201
+ "rewards/env_goofspiel_reward/std": 0.39523468613624574,
202
+ "sampling/importance_sampling_ratio/max": 2.1884907484054565,
203
+ "sampling/importance_sampling_ratio/mean": 0.9641352295875549,
204
+ "sampling/importance_sampling_ratio/min": 0.17558300793170928,
205
+ "sampling/sampling_logp_difference/max": 1.910474991798401,
206
+ "sampling/sampling_logp_difference/mean": 0.11390596330165863,
207
  "step": 30,
208
+ "step_time": 5.248301958399679
209
  },
210
  {
211
+ "clip_ratio/high_max": 0.021911457646638155,
212
+ "clip_ratio/high_mean": 0.010955728823319077,
213
+ "clip_ratio/low_mean": 0.016263545025140047,
214
+ "clip_ratio/low_min": 0.002631578966975212,
215
+ "clip_ratio/region_mean": 0.02721927403472364,
216
  "completions/clipped_ratio": 0.0,
217
+ "completions/max_length": 366.0,
218
+ "completions/max_terminated_length": 366.0,
219
+ "completions/mean_length": 290.6625,
220
+ "completions/mean_terminated_length": 290.6625,
221
  "completions/min_length": 212.0,
222
  "completions/min_terminated_length": 212.0,
223
+ "entropy": 0.4240268304944038,
224
+ "epoch": 0.0056,
225
+ "frac_reward_zero_std": 0.4,
226
+ "grad_norm": 0.08057750761508942,
227
+ "kl": 1.8736468333750964,
228
  "learning_rate": 9.666336e-06,
229
+ "loss": -2.150831278413534e-05,
230
+ "num_tokens": 940033.0,
231
+ "reward": 0.4274375081062317,
232
+ "reward_std": 0.2758600294589996,
233
+ "rewards/env_goofspiel_reward/mean": 0.4274375081062317,
234
+ "rewards/env_goofspiel_reward/std": 0.4116846978664398,
235
+ "sampling/importance_sampling_ratio/max": 2.5006643772125243,
236
+ "sampling/importance_sampling_ratio/mean": 0.966198992729187,
237
+ "sampling/importance_sampling_ratio/min": 0.07496144040487707,
238
+ "sampling/sampling_logp_difference/max": 2.63707594871521,
239
+ "sampling/sampling_logp_difference/mean": 0.1406691253185272,
240
  "step": 35,
241
+ "step_time": 5.299385342999813
242
  },
243
  {
244
+ "clip_ratio/high_max": 0.020319487527012826,
245
+ "clip_ratio/high_mean": 0.010159743763506413,
246
+ "clip_ratio/low_mean": 0.007234477158635855,
247
  "clip_ratio/low_min": 0.0,
248
+ "clip_ratio/region_mean": 0.017394221015274526,
249
  "completions/clipped_ratio": 0.0,
250
+ "completions/max_length": 374.2,
251
+ "completions/max_terminated_length": 374.2,
252
+ "completions/mean_length": 289.475,
253
+ "completions/mean_terminated_length": 289.475,
254
+ "completions/min_length": 207.0,
255
+ "completions/min_terminated_length": 207.0,
256
+ "entropy": 0.6166644155979156,
257
+ "epoch": 0.0064,
258
+ "frac_reward_zero_std": 0.5,
259
+ "grad_norm": 0.051475733518600464,
260
+ "kl": 0.6312531501054763,
261
+ "learning_rate": 9.95063915881342e-06,
262
+ "loss": 0.0008587016724050045,
263
+ "num_tokens": 1074989.0,
264
+ "reward": 0.2586875051259995,
265
+ "reward_std": 0.2599501311779022,
266
+ "rewards/env_goofspiel_reward/mean": 0.2586875051259995,
267
+ "rewards/env_goofspiel_reward/std": 0.38087824583053587,
268
+ "sampling/importance_sampling_ratio/max": 2.293054127693176,
269
+ "sampling/importance_sampling_ratio/mean": 0.9133782982826233,
270
+ "sampling/importance_sampling_ratio/min": 0.09586721286177635,
271
+ "sampling/sampling_logp_difference/max": 1.7645570278167724,
272
+ "sampling/sampling_logp_difference/mean": 0.1391677066683769,
273
  "step": 40,
274
+ "step_time": 5.214609204999943
275
  },
276
  {
277
+ "clip_ratio/high_max": 0.01441670972853899,
278
+ "clip_ratio/high_mean": 0.007208354864269495,
279
+ "clip_ratio/low_mean": 0.005737766716629266,
280
+ "clip_ratio/low_min": 0.002631578966975212,
281
+ "clip_ratio/region_mean": 0.012946121580898761,
282
  "completions/clipped_ratio": 0.0,
283
+ "completions/max_length": 374.2,
284
+ "completions/max_terminated_length": 374.2,
285
+ "completions/mean_length": 294.13125,
286
+ "completions/mean_terminated_length": 294.13125,
287
+ "completions/min_length": 184.6,
288
+ "completions/min_terminated_length": 184.6,
289
+ "entropy": 0.7186032980680466,
290
+ "epoch": 0.0072,
291
+ "frac_reward_zero_std": 0.5125,
292
+ "grad_norm": 0.10908176004886627,
293
+ "kl": 0.9199723824858665,
294
+ "learning_rate": 9.950635741493589e-06,
295
+ "loss": 0.0010974571108818055,
296
+ "num_tokens": 1211700.0,
297
+ "reward": 0.20568750202655792,
298
+ "reward_std": 0.21823083460330964,
299
+ "rewards/env_goofspiel_reward/mean": 0.20568750202655792,
300
+ "rewards/env_goofspiel_reward/std": 0.3572053849697113,
301
+ "sampling/importance_sampling_ratio/max": 2.1740296363830565,
302
+ "sampling/importance_sampling_ratio/mean": 0.8587659239768982,
303
+ "sampling/importance_sampling_ratio/min": 0.19039739817380905,
304
+ "sampling/sampling_logp_difference/max": 1.302869963645935,
305
+ "sampling/sampling_logp_difference/mean": 0.159588959813118,
306
  "step": 45,
307
+ "step_time": 5.3400724014001755
308
  },
309
  {
310
+ "clip_ratio/high_max": 0.014580108411610126,
311
+ "clip_ratio/high_mean": 0.008760642446577548,
312
+ "clip_ratio/low_mean": 0.005718954280018807,
313
+ "clip_ratio/low_min": 0.002777777798473835,
314
+ "clip_ratio/region_mean": 0.014479596912860871,
315
  "completions/clipped_ratio": 0.0,
316
+ "completions/max_length": 374.0,
317
+ "completions/max_terminated_length": 374.0,
318
+ "completions/mean_length": 298.4875,
319
+ "completions/mean_terminated_length": 298.4875,
320
+ "completions/min_length": 194.6,
321
+ "completions/min_terminated_length": 194.6,
322
+ "entropy": 0.7520321547985077,
323
+ "epoch": 0.008,
324
+ "frac_reward_zero_std": 0.55,
325
+ "grad_norm": 0.07008689641952515,
326
+ "kl": 1.4948164954781533,
327
+ "learning_rate": 9.950629695468755e-06,
328
+ "loss": 0.000722643407061696,
329
+ "num_tokens": 1348707.0,
330
+ "reward": 0.19093750715255736,
331
+ "reward_std": 0.18605746924877167,
332
+ "rewards/env_goofspiel_reward/mean": 0.19093750715255736,
333
+ "rewards/env_goofspiel_reward/std": 0.32617470622062683,
334
+ "sampling/importance_sampling_ratio/max": 2.441471576690674,
335
+ "sampling/importance_sampling_ratio/mean": 0.8815865159034729,
336
+ "sampling/importance_sampling_ratio/min": 0.047414033114910124,
337
+ "sampling/sampling_logp_difference/max": 1.4149149417877198,
338
+ "sampling/sampling_logp_difference/mean": 0.17731134295463563,
339
  "step": 50,
340
+ "step_time": 5.3226749666000615
341
  },
342
  {
343
+ "clip_ratio/high_max": 0.017320261523127555,
344
+ "clip_ratio/high_mean": 0.008660130761563778,
345
+ "clip_ratio/low_mean": 0.0015625,
346
  "clip_ratio/low_min": 0.0,
347
+ "clip_ratio/region_mean": 0.010222630761563777,
348
  "completions/clipped_ratio": 0.0,
349
+ "completions/max_length": 373.6,
350
+ "completions/max_terminated_length": 373.6,
351
+ "completions/mean_length": 284.1125,
352
+ "completions/mean_terminated_length": 284.1125,
353
+ "completions/min_length": 200.0,
354
+ "completions/min_terminated_length": 200.0,
355
+ "entropy": 0.7574372291564941,
356
+ "epoch": 0.0088,
357
+ "frac_reward_zero_std": 0.6125,
358
+ "grad_norm": 0.05677078291773796,
359
+ "kl": 0.9213189110159874,
360
+ "learning_rate": 9.950621020743173e-06,
361
+ "loss": 0.00019418969750404358,
362
+ "num_tokens": 1481168.0,
363
+ "reward": 0.16468750387430192,
364
+ "reward_std": 0.15954096913337706,
365
+ "rewards/env_goofspiel_reward/mean": 0.16468750387430192,
366
+ "rewards/env_goofspiel_reward/std": 0.3025706380605698,
367
+ "sampling/importance_sampling_ratio/max": 2.087395262718201,
368
+ "sampling/importance_sampling_ratio/mean": 0.9323906660079956,
369
+ "sampling/importance_sampling_ratio/min": 0.12477300018072128,
370
+ "sampling/sampling_logp_difference/max": 1.5357290506362915,
371
+ "sampling/sampling_logp_difference/mean": 0.16227305233478545,
372
  "step": 55,
373
+ "step_time": 5.239984228400317
374
  },
375
  {
376
+ "clip_ratio/high_max": 0.011312134563922882,
377
+ "clip_ratio/high_mean": 0.005656067281961441,
378
+ "clip_ratio/low_mean": 0.008540054224431515,
379
  "clip_ratio/low_min": 0.0,
380
+ "clip_ratio/region_mean": 0.014196121599525213,
381
+ "completions/clipped_ratio": 0.0,
382
+ "completions/max_length": 374.8,
383
+ "completions/max_terminated_length": 374.8,
384
+ "completions/mean_length": 294.125,
385
+ "completions/mean_terminated_length": 294.125,
386
+ "completions/min_length": 218.4,
387
+ "completions/min_terminated_length": 218.4,
388
+ "entropy": 0.7390435010194778,
389
+ "epoch": 0.0096,
390
+ "frac_reward_zero_std": 0.5,
391
+ "grad_norm": 0.059227459132671356,
392
+ "kl": 0.8731714501976967,
393
+ "learning_rate": 9.950609717322956e-06,
394
+ "loss": 0.00016935726162046195,
395
+ "num_tokens": 1616969.0,
396
+ "reward": 0.22868750393390655,
397
+ "reward_std": 0.23873692452907563,
398
+ "rewards/env_goofspiel_reward/mean": 0.22868750393390655,
399
+ "rewards/env_goofspiel_reward/std": 0.3541231632232666,
400
+ "sampling/importance_sampling_ratio/max": 2.4580262422561647,
401
+ "sampling/importance_sampling_ratio/mean": 1.027009415626526,
402
+ "sampling/importance_sampling_ratio/min": 0.08994593024253845,
403
+ "sampling/sampling_logp_difference/max": 1.2262043237686158,
404
+ "sampling/sampling_logp_difference/mean": 0.15936054587364196,
405
+ "step": 60,
406
+ "step_time": 5.197383608799828
407
+ },
408
+ {
409
+ "clip_ratio/high_max": 0.003125,
410
+ "clip_ratio/high_mean": 0.0015625,
411
+ "clip_ratio/low_mean": 0.004093567281961441,
412
+ "clip_ratio/low_min": 0.0,
413
+ "clip_ratio/region_mean": 0.005656067281961441,
414
  "completions/clipped_ratio": 0.0,
415
  "completions/max_length": 374.0,
416
  "completions/max_terminated_length": 374.0,
417
+ "completions/mean_length": 272.7125,
418
+ "completions/mean_terminated_length": 272.7125,
419
+ "completions/min_length": 207.0,
420
+ "completions/min_terminated_length": 207.0,
421
+ "entropy": 0.641479243338108,
422
+ "epoch": 0.0104,
423
+ "frac_reward_zero_std": 0.5125,
424
+ "grad_norm": 0.052345700562000275,
425
+ "kl": 0.8409833669662475,
426
+ "learning_rate": 9.950595785216067e-06,
427
+ "loss": -0.0004354896955192089,
428
+ "num_tokens": 1745735.0,
429
+ "reward": 0.30350000858306886,
430
+ "reward_std": 0.24943190813064575,
431
+ "rewards/env_goofspiel_reward/mean": 0.30350000858306886,
432
+ "rewards/env_goofspiel_reward/std": 0.43262303471565244,
433
+ "sampling/importance_sampling_ratio/max": 2.308809924125671,
434
+ "sampling/importance_sampling_ratio/mean": 0.9729154109954834,
435
+ "sampling/importance_sampling_ratio/min": 0.2105877071619034,
436
+ "sampling/sampling_logp_difference/max": 1.216427493095398,
437
+ "sampling/sampling_logp_difference/mean": 0.14146182239055632,
438
+ "step": 65,
439
+ "step_time": 5.165435432399863
440
  },
441
  {
442
+ "clip_ratio/high_max": 0.008823529444634914,
443
+ "clip_ratio/high_mean": 0.004411764722317457,
444
+ "clip_ratio/low_mean": 0.004421977140009403,
445
  "clip_ratio/low_min": 0.0,
446
+ "clip_ratio/region_mean": 0.00883374186232686,
447
  "completions/clipped_ratio": 0.0,
448
  "completions/max_length": 374.0,
449
  "completions/max_terminated_length": 374.0,
450
+ "completions/mean_length": 279.70625,
451
+ "completions/mean_terminated_length": 279.70625,
452
+ "completions/min_length": 194.6,
453
+ "completions/min_terminated_length": 194.6,
454
+ "entropy": 0.5012152880430222,
455
+ "epoch": 0.0112,
456
+ "frac_reward_zero_std": 0.325,
457
+ "grad_norm": 0.08534521609544754,
458
+ "kl": 0.8132658362388611,
459
+ "learning_rate": 9.950579224432321e-06,
460
+ "loss": -0.0004163400735706091,
461
+ "num_tokens": 1877877.0,
462
+ "reward": 0.4497500121593475,
463
+ "reward_std": 0.32915821075439455,
464
+ "rewards/env_goofspiel_reward/mean": 0.4497500121593475,
465
+ "rewards/env_goofspiel_reward/std": 0.4289704501628876,
466
+ "sampling/importance_sampling_ratio/max": 2.469445323944092,
467
+ "sampling/importance_sampling_ratio/mean": 0.9713802933692932,
468
+ "sampling/importance_sampling_ratio/min": 0.1933848649263382,
469
+ "sampling/sampling_logp_difference/max": 1.2900412559509278,
470
+ "sampling/sampling_logp_difference/mean": 0.12238389104604722,
471
+ "step": 70,
472
+ "step_time": 5.296977608999805
473
+ },
474
+ {
475
+ "clip_ratio/high_max": 0.005409356765449047,
476
+ "clip_ratio/high_mean": 0.0027046783827245234,
477
+ "clip_ratio/low_mean": 0.011513618659228087,
478
+ "clip_ratio/low_min": 0.0,
479
+ "clip_ratio/region_mean": 0.01421829704195261,
480
+ "completions/clipped_ratio": 0.0,
481
+ "completions/max_length": 374.4,
482
+ "completions/max_terminated_length": 374.4,
483
+ "completions/mean_length": 292.43125,
484
+ "completions/mean_terminated_length": 292.43125,
485
  "completions/min_length": 212.0,
486
  "completions/min_terminated_length": 212.0,
487
+ "entropy": 0.39477833807468415,
488
+ "epoch": 0.012,
489
+ "frac_reward_zero_std": 0.45,
490
+ "grad_norm": 0.08385973423719406,
491
+ "kl": 0.9824723288416862,
492
+ "learning_rate": 9.950560034983382e-06,
493
+ "loss": -0.0006953636649996043,
494
+ "num_tokens": 2012929.0,
495
+ "reward": 0.5398125350475311,
496
+ "reward_std": 0.2548236042261124,
497
+ "rewards/env_goofspiel_reward/mean": 0.5398125350475311,
498
+ "rewards/env_goofspiel_reward/std": 0.4067754566669464,
499
+ "sampling/importance_sampling_ratio/max": 2.476490020751953,
500
+ "sampling/importance_sampling_ratio/mean": 0.9818659067153931,
501
+ "sampling/importance_sampling_ratio/min": 0.04677087515592575,
502
+ "sampling/sampling_logp_difference/max": 1.8360216617584229,
503
+ "sampling/sampling_logp_difference/mean": 0.1339985266327858,
504
+ "step": 75,
505
+ "step_time": 5.25195668339984
506
+ },
507
+ {
508
+ "epoch": 0.012,
509
+ "eval_clip_ratio/high_max": 0.0,
510
+ "eval_clip_ratio/high_mean": 0.0,
511
+ "eval_clip_ratio/low_mean": 0.0,
512
+ "eval_clip_ratio/low_min": 0.0,
513
+ "eval_clip_ratio/region_mean": 0.0,
514
+ "eval_completions/clipped_ratio": 0.0,
515
+ "eval_completions/max_length": 310.0,
516
+ "eval_completions/max_terminated_length": 310.0,
517
+ "eval_completions/mean_length": 274.9,
518
+ "eval_completions/mean_terminated_length": 274.9,
519
+ "eval_completions/min_length": 241.2,
520
+ "eval_completions/min_terminated_length": 241.2,
521
+ "eval_entropy": 0.2987076103687286,
522
+ "eval_frac_reward_zero_std": 0.6,
523
+ "eval_kl": 0.8131496548652649,
524
+ "eval_loss": -0.0002675331197679043,
525
+ "eval_num_tokens": 2012929.0,
526
+ "eval_reward": 0.5700000166893006,
527
+ "eval_reward_std": 0.2121320366859436,
528
+ "eval_rewards/env_goofspiel_reward/mean": 0.5700000166893006,
529
+ "eval_rewards/env_goofspiel_reward/std": 0.3252412259578705,
530
+ "eval_runtime": 2.5856,
531
+ "eval_samples_per_second": 3.868,
532
+ "eval_sampling/importance_sampling_ratio/max": 1.6977968692779541,
533
+ "eval_sampling/importance_sampling_ratio/mean": 0.947873055934906,
534
+ "eval_sampling/importance_sampling_ratio/min": 0.2541299909353256,
535
+ "eval_sampling/sampling_logp_difference/max": 1.3066397070884705,
536
+ "eval_sampling/sampling_logp_difference/mean": 0.14956869557499886,
537
+ "eval_steps_per_second": 1.16,
538
+ "step": 75
539
  }
540
  ],
541
  "logging_steps": 5,
542
+ "max_steps": 18750,
543
+ "num_input_tokens_seen": 2012929,
544
  "num_train_epochs": 3,
545
  "save_steps": 500,
546
  "stateful_callbacks": {
 
550
  "should_evaluate": false,
551
  "should_log": false,
552
  "should_save": true,
553
+ "should_training_stop": false
554
  },
555
  "attributes": {}
556
  }
557
  },
558
  "total_flos": 0.0,
559
+ "train_batch_size": 4,
560
  "trial_name": null,
561
  "trial_params": null
562
  }
training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3300a89e75ebdeb9256a0a6e1d02dfcc2078fc90cd06fa1740dbf57e59261452
3
  size 7185
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:906bc07f18d85f3fdbe47d01e60bbe6f967852d19caecc88d502ce07c5e4aa78
3
  size 7185