CodeIsAbstract commited on
Commit
bfd8f27
·
verified ·
1 Parent(s): 45bef51

Training in progress, step 32000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:78d2e42563ecfd65d97699a72f4be839c4815a628d9aafa31de38f84519e5ebf
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fc3c5b4f1c32d282710a153ae5bf5ca6bd7cbd3f4f67f12a3907a6cd19d2d279
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d6cf00bdfb1404208c2471c83afcc106348ceb61f5610d2591e45fff9be83487
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:605c9b21acde75c1394f97c25a318d3192f035d829b96871f4ced0b57206ec1c
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:09745e8006c686b7ca1b3f3ce96dc9d5406d13f67ce84984ece76a3860e0757b
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6177091d3805fe7255fbeea64136d688b15629fc946401f1853fbd6e6a2f18ad
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1a7a45fe9f879fdefbf9d22a9cd4381d35f0cc9247510e28a7632ac32b2c74e6
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:30eae1c116a1408bbe4aec95a1368b13c0bf69d349ca373bf4a8fd3a1e9fc110
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.2545454545454545,
6
  "eval_steps": 1000,
7
- "global_step": 28000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2192,6 +2192,318 @@
2192
  "eval_samples_per_second": 84.778,
2193
  "eval_steps_per_second": 21.195,
2194
  "step": 28000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2195
  }
2196
  ],
2197
  "logging_steps": 100,
@@ -2211,7 +2523,7 @@
2211
  "attributes": {}
2212
  }
2213
  },
2214
- "total_flos": 6.97608267890688e+17,
2215
  "train_batch_size": 22,
2216
  "trial_name": null,
2217
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.2909090909090909,
6
  "eval_steps": 1000,
7
+ "global_step": 32000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2192
  "eval_samples_per_second": 84.778,
2193
  "eval_steps_per_second": 21.195,
2194
  "step": 28000
2195
+ },
2196
+ {
2197
+ "epoch": 0.25545454545454543,
2198
+ "grad_norm": 0.1713736206293106,
2199
+ "learning_rate": 0.0008486002062601203,
2200
+ "loss": 2.8171258544921876,
2201
+ "step": 28100
2202
+ },
2203
+ {
2204
+ "epoch": 0.25636363636363635,
2205
+ "grad_norm": 0.18038378655910492,
2206
+ "learning_rate": 0.000847573687285321,
2207
+ "loss": 2.8066680908203123,
2208
+ "step": 28200
2209
+ },
2210
+ {
2211
+ "epoch": 0.25727272727272726,
2212
+ "grad_norm": 0.1601206362247467,
2213
+ "learning_rate": 0.0008465443255111025,
2214
+ "loss": 2.8091595458984373,
2215
+ "step": 28300
2216
+ },
2217
+ {
2218
+ "epoch": 0.2581818181818182,
2219
+ "grad_norm": 0.15122343599796295,
2220
+ "learning_rate": 0.0008455121293565974,
2221
+ "loss": 2.8354150390625,
2222
+ "step": 28400
2223
+ },
2224
+ {
2225
+ "epoch": 0.2590909090909091,
2226
+ "grad_norm": 0.1411023885011673,
2227
+ "learning_rate": 0.000844477107264121,
2228
+ "loss": 2.805926513671875,
2229
+ "step": 28500
2230
+ },
2231
+ {
2232
+ "epoch": 0.26,
2233
+ "grad_norm": 0.1714688241481781,
2234
+ "learning_rate": 0.0008434392676991018,
2235
+ "loss": 2.774625244140625,
2236
+ "step": 28600
2237
+ },
2238
+ {
2239
+ "epoch": 0.2609090909090909,
2240
+ "grad_norm": 0.14336396753787994,
2241
+ "learning_rate": 0.0008423986191500126,
2242
+ "loss": 2.806527099609375,
2243
+ "step": 28700
2244
+ },
2245
+ {
2246
+ "epoch": 0.26181818181818184,
2247
+ "grad_norm": 0.14724823832511902,
2248
+ "learning_rate": 0.0008413551701283003,
2249
+ "loss": 2.7858685302734374,
2250
+ "step": 28800
2251
+ },
2252
+ {
2253
+ "epoch": 0.26272727272727275,
2254
+ "grad_norm": 0.1667833775281906,
2255
+ "learning_rate": 0.0008403089291683173,
2256
+ "loss": 2.788014221191406,
2257
+ "step": 28900
2258
+ },
2259
+ {
2260
+ "epoch": 0.2636363636363636,
2261
+ "grad_norm": 0.13815470039844513,
2262
+ "learning_rate": 0.0008392599048272509,
2263
+ "loss": 2.7937637329101563,
2264
+ "step": 29000
2265
+ },
2266
+ {
2267
+ "epoch": 0.2636363636363636,
2268
+ "eval_loss": 3.1548070907592773,
2269
+ "eval_runtime": 6.8019,
2270
+ "eval_samples_per_second": 84.682,
2271
+ "eval_steps_per_second": 21.171,
2272
+ "step": 29000
2273
+ },
2274
+ {
2275
+ "epoch": 0.26454545454545453,
2276
+ "grad_norm": 0.1442345827817917,
2277
+ "learning_rate": 0.0008382081056850541,
2278
+ "loss": 2.77263671875,
2279
+ "step": 29100
2280
+ },
2281
+ {
2282
+ "epoch": 0.26545454545454544,
2283
+ "grad_norm": 0.14780735969543457,
2284
+ "learning_rate": 0.0008371535403443743,
2285
+ "loss": 2.8082760620117186,
2286
+ "step": 29200
2287
+ },
2288
+ {
2289
+ "epoch": 0.26636363636363636,
2290
+ "grad_norm": 0.13575103878974915,
2291
+ "learning_rate": 0.000836096217430484,
2292
+ "loss": 2.7685586547851564,
2293
+ "step": 29300
2294
+ },
2295
+ {
2296
+ "epoch": 0.2672727272727273,
2297
+ "grad_norm": 0.131728395819664,
2298
+ "learning_rate": 0.0008350361455912099,
2299
+ "loss": 2.771170349121094,
2300
+ "step": 29400
2301
+ },
2302
+ {
2303
+ "epoch": 0.2681818181818182,
2304
+ "grad_norm": 0.18209482729434967,
2305
+ "learning_rate": 0.0008339733334968619,
2306
+ "loss": 2.7749264526367186,
2307
+ "step": 29500
2308
+ },
2309
+ {
2310
+ "epoch": 0.2690909090909091,
2311
+ "grad_norm": 0.14567936956882477,
2312
+ "learning_rate": 0.0008329077898401624,
2313
+ "loss": 2.8004241943359376,
2314
+ "step": 29600
2315
+ },
2316
+ {
2317
+ "epoch": 0.27,
2318
+ "grad_norm": 0.1705045998096466,
2319
+ "learning_rate": 0.0008318395233361753,
2320
+ "loss": 2.8070437622070314,
2321
+ "step": 29700
2322
+ },
2323
+ {
2324
+ "epoch": 0.27090909090909093,
2325
+ "grad_norm": 0.1700645238161087,
2326
+ "learning_rate": 0.0008307685427222345,
2327
+ "loss": 2.769403991699219,
2328
+ "step": 29800
2329
+ },
2330
+ {
2331
+ "epoch": 0.2718181818181818,
2332
+ "grad_norm": 0.2213045209646225,
2333
+ "learning_rate": 0.000829694856757873,
2334
+ "loss": 2.75067138671875,
2335
+ "step": 29900
2336
+ },
2337
+ {
2338
+ "epoch": 0.2727272727272727,
2339
+ "grad_norm": 0.1627679318189621,
2340
+ "learning_rate": 0.0008286184742247502,
2341
+ "loss": 2.782334899902344,
2342
+ "step": 30000
2343
+ },
2344
+ {
2345
+ "epoch": 0.2727272727272727,
2346
+ "eval_loss": 3.1495797634124756,
2347
+ "eval_runtime": 6.8699,
2348
+ "eval_samples_per_second": 83.844,
2349
+ "eval_steps_per_second": 20.961,
2350
+ "step": 30000
2351
+ },
2352
+ {
2353
+ "epoch": 0.2736363636363636,
2354
+ "grad_norm": 0.1445474475622177,
2355
+ "learning_rate": 0.0008275394039265813,
2356
+ "loss": 2.80556396484375,
2357
+ "step": 30100
2358
+ },
2359
+ {
2360
+ "epoch": 0.27454545454545454,
2361
+ "grad_norm": 0.14685708284378052,
2362
+ "learning_rate": 0.0008264576546890639,
2363
+ "loss": 2.776888732910156,
2364
+ "step": 30200
2365
+ },
2366
+ {
2367
+ "epoch": 0.27545454545454545,
2368
+ "grad_norm": 0.1347658634185791,
2369
+ "learning_rate": 0.0008253732353598072,
2370
+ "loss": 2.7627569580078126,
2371
+ "step": 30300
2372
+ },
2373
+ {
2374
+ "epoch": 0.27636363636363637,
2375
+ "grad_norm": 0.13901016116142273,
2376
+ "learning_rate": 0.0008242861548082591,
2377
+ "loss": 2.780741882324219,
2378
+ "step": 30400
2379
+ },
2380
+ {
2381
+ "epoch": 0.2772727272727273,
2382
+ "grad_norm": 0.13585121929645538,
2383
+ "learning_rate": 0.0008231964219256331,
2384
+ "loss": 2.804005432128906,
2385
+ "step": 30500
2386
+ },
2387
+ {
2388
+ "epoch": 0.2781818181818182,
2389
+ "grad_norm": 0.15015123784542084,
2390
+ "learning_rate": 0.0008221040456248367,
2391
+ "loss": 2.7642987060546873,
2392
+ "step": 30600
2393
+ },
2394
+ {
2395
+ "epoch": 0.2790909090909091,
2396
+ "grad_norm": 0.13883176445960999,
2397
+ "learning_rate": 0.0008210090348403973,
2398
+ "loss": 2.79775146484375,
2399
+ "step": 30700
2400
+ },
2401
+ {
2402
+ "epoch": 0.28,
2403
+ "grad_norm": 0.1524442732334137,
2404
+ "learning_rate": 0.0008199113985283902,
2405
+ "loss": 2.7866134643554688,
2406
+ "step": 30800
2407
+ },
2408
+ {
2409
+ "epoch": 0.2809090909090909,
2410
+ "grad_norm": 0.14497579634189606,
2411
+ "learning_rate": 0.0008188111456663641,
2412
+ "loss": 2.770282897949219,
2413
+ "step": 30900
2414
+ },
2415
+ {
2416
+ "epoch": 0.2818181818181818,
2417
+ "grad_norm": 0.1663127839565277,
2418
+ "learning_rate": 0.0008177082852532691,
2419
+ "loss": 2.7479037475585937,
2420
+ "step": 31000
2421
+ },
2422
+ {
2423
+ "epoch": 0.2818181818181818,
2424
+ "eval_loss": 3.1455507278442383,
2425
+ "eval_runtime": 6.8464,
2426
+ "eval_samples_per_second": 84.131,
2427
+ "eval_steps_per_second": 21.033,
2428
+ "step": 31000
2429
+ },
2430
+ {
2431
+ "epoch": 0.2827272727272727,
2432
+ "grad_norm": 0.3059700131416321,
2433
+ "learning_rate": 0.0008166028263093825,
2434
+ "loss": 2.786407775878906,
2435
+ "step": 31100
2436
+ },
2437
+ {
2438
+ "epoch": 0.28363636363636363,
2439
+ "grad_norm": 0.1757507622241974,
2440
+ "learning_rate": 0.0008154947778762343,
2441
+ "loss": 2.7669387817382813,
2442
+ "step": 31200
2443
+ },
2444
+ {
2445
+ "epoch": 0.28454545454545455,
2446
+ "grad_norm": 0.1763867884874344,
2447
+ "learning_rate": 0.0008143841490165344,
2448
+ "loss": 2.763365478515625,
2449
+ "step": 31300
2450
+ },
2451
+ {
2452
+ "epoch": 0.28545454545454546,
2453
+ "grad_norm": 0.1497093141078949,
2454
+ "learning_rate": 0.0008132709488140977,
2455
+ "loss": 2.7558905029296876,
2456
+ "step": 31400
2457
+ },
2458
+ {
2459
+ "epoch": 0.2863636363636364,
2460
+ "grad_norm": 0.15906350314617157,
2461
+ "learning_rate": 0.0008121551863737704,
2462
+ "loss": 2.7421307373046875,
2463
+ "step": 31500
2464
+ },
2465
+ {
2466
+ "epoch": 0.2872727272727273,
2467
+ "grad_norm": 0.14353445172309875,
2468
+ "learning_rate": 0.0008110368708213548,
2469
+ "loss": 2.7798092651367186,
2470
+ "step": 31600
2471
+ },
2472
+ {
2473
+ "epoch": 0.2881818181818182,
2474
+ "grad_norm": 0.1670762598514557,
2475
+ "learning_rate": 0.0008099160113035353,
2476
+ "loss": 2.764653625488281,
2477
+ "step": 31700
2478
+ },
2479
+ {
2480
+ "epoch": 0.28909090909090907,
2481
+ "grad_norm": 0.15554283559322357,
2482
+ "learning_rate": 0.000808792616987803,
2483
+ "loss": 2.76354736328125,
2484
+ "step": 31800
2485
+ },
2486
+ {
2487
+ "epoch": 0.29,
2488
+ "grad_norm": 0.14061635732650757,
2489
+ "learning_rate": 0.0008076666970623821,
2490
+ "loss": 2.764184875488281,
2491
+ "step": 31900
2492
+ },
2493
+ {
2494
+ "epoch": 0.2909090909090909,
2495
+ "grad_norm": 0.18086077272891998,
2496
+ "learning_rate": 0.0008065382607361523,
2497
+ "loss": 2.7778125,
2498
+ "step": 32000
2499
+ },
2500
+ {
2501
+ "epoch": 0.2909090909090909,
2502
+ "eval_loss": 3.1462559700012207,
2503
+ "eval_runtime": 6.8321,
2504
+ "eval_samples_per_second": 84.308,
2505
+ "eval_steps_per_second": 21.077,
2506
+ "step": 32000
2507
  }
2508
  ],
2509
  "logging_steps": 100,
 
2523
  "attributes": {}
2524
  }
2525
  },
2526
+ "total_flos": 7.97266591875072e+17,
2527
  "train_batch_size": 22,
2528
  "trial_name": null,
2529
  "trial_params": null