CodeIsAbstract commited on
Commit
6b12a8f
·
verified ·
1 Parent(s): 6ca6c44

Training in progress, step 6000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:93b4d618fee0aaf987ff44511c965a4a13d55a3d93f354ed73c1f91c3e21dccc
3
  size 496262784
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9a3b1570476040558106ef329ac5c35eaf7b927d872082a820226ad0028b5dd9
3
  size 496262784
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d79704be2307835f0582a7f7abe29d698ca7dbc76fdb78dc7bbfd0aaa6990b77
3
  size 992621963
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ac1c439f2b333c6ba22b924fd6eb4e28a8524e3ac41694760d6b949858f9f4ad
3
  size 992621963
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c9e2628e78d1dc996f476f66725a5d868b6e07dce08fc0df5d5fb50fcc8f9636
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:595e0adb3f51b4014c388d14cb8a20e5767cd34a947ffb570078cd186928d2c7
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c683ea10141469aa50dfa283ef5ec8ac410a06cef28bfb1e66e6a7d94e82730d
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f43de46bf0618be59b12a78ea440d0013f604015e857ea4a03ea845dabb299ad
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.08,
6
  "eval_steps": 1000,
7
- "global_step": 4000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -324,6 +324,164 @@
324
  "eval_samples_per_second": 309.393,
325
  "eval_steps_per_second": 19.447,
326
  "step": 4000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
327
  }
328
  ],
329
  "logging_steps": 100,
@@ -343,7 +501,7 @@
343
  "attributes": {}
344
  }
345
  },
346
- "total_flos": 1.2542017536e+17,
347
  "train_batch_size": 120,
348
  "trial_name": null,
349
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.12,
6
  "eval_steps": 1000,
7
+ "global_step": 6000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
324
  "eval_samples_per_second": 309.393,
325
  "eval_steps_per_second": 19.447,
326
  "step": 4000
327
+ },
328
+ {
329
+ "epoch": 0.082,
330
+ "grad_norm": 0.2674967348575592,
331
+ "learning_rate": 0.0005907574406673935,
332
+ "loss": 3.9582391357421876,
333
+ "step": 4100
334
+ },
335
+ {
336
+ "epoch": 0.084,
337
+ "grad_norm": 0.25437116622924805,
338
+ "learning_rate": 0.0005902859921636611,
339
+ "loss": 3.942552185058594,
340
+ "step": 4200
341
+ },
342
+ {
343
+ "epoch": 0.086,
344
+ "grad_norm": 0.25480520725250244,
345
+ "learning_rate": 0.0005898030145956102,
346
+ "loss": 3.9550421142578127,
347
+ "step": 4300
348
+ },
349
+ {
350
+ "epoch": 0.088,
351
+ "grad_norm": 0.25695815682411194,
352
+ "learning_rate": 0.0005893085271452874,
353
+ "loss": 3.9156878662109373,
354
+ "step": 4400
355
+ },
356
+ {
357
+ "epoch": 0.09,
358
+ "grad_norm": 0.25535979866981506,
359
+ "learning_rate": 0.0005888025494518687,
360
+ "loss": 3.9294302368164065,
361
+ "step": 4500
362
+ },
363
+ {
364
+ "epoch": 0.092,
365
+ "grad_norm": 0.24557402729988098,
366
+ "learning_rate": 0.0005882851016108786,
367
+ "loss": 3.927276306152344,
368
+ "step": 4600
369
+ },
370
+ {
371
+ "epoch": 0.094,
372
+ "grad_norm": 0.26716557145118713,
373
+ "learning_rate": 0.0005877562041733932,
374
+ "loss": 3.904244384765625,
375
+ "step": 4700
376
+ },
377
+ {
378
+ "epoch": 0.096,
379
+ "grad_norm": 0.2525787651538849,
380
+ "learning_rate": 0.0005872158781452231,
381
+ "loss": 3.8892666625976564,
382
+ "step": 4800
383
+ },
384
+ {
385
+ "epoch": 0.098,
386
+ "grad_norm": 0.3125140964984894,
387
+ "learning_rate": 0.0005866641449860792,
388
+ "loss": 3.887486572265625,
389
+ "step": 4900
390
+ },
391
+ {
392
+ "epoch": 0.1,
393
+ "grad_norm": 0.2451365739107132,
394
+ "learning_rate": 0.0005861010266087211,
395
+ "loss": 3.875788269042969,
396
+ "step": 5000
397
+ },
398
+ {
399
+ "epoch": 0.1,
400
+ "eval_accuracy": 0.32875099183244255,
401
+ "eval_loss": 3.831427812576294,
402
+ "eval_runtime": 5.9329,
403
+ "eval_samples_per_second": 327.159,
404
+ "eval_steps_per_second": 20.563,
405
+ "step": 5000
406
+ },
407
+ {
408
+ "epoch": 0.102,
409
+ "grad_norm": 0.28080520033836365,
410
+ "learning_rate": 0.0005855265453780858,
411
+ "loss": 3.8711001586914064,
412
+ "step": 5100
413
+ },
414
+ {
415
+ "epoch": 0.104,
416
+ "grad_norm": 0.25133365392684937,
417
+ "learning_rate": 0.0005849407241104003,
418
+ "loss": 3.8725830078125,
419
+ "step": 5200
420
+ },
421
+ {
422
+ "epoch": 0.106,
423
+ "grad_norm": 0.27164554595947266,
424
+ "learning_rate": 0.000584343586072275,
425
+ "loss": 3.847073974609375,
426
+ "step": 5300
427
+ },
428
+ {
429
+ "epoch": 0.108,
430
+ "grad_norm": 0.2697848081588745,
431
+ "learning_rate": 0.0005837351549797795,
432
+ "loss": 3.79802978515625,
433
+ "step": 5400
434
+ },
435
+ {
436
+ "epoch": 0.11,
437
+ "grad_norm": 0.27253684401512146,
438
+ "learning_rate": 0.0005831154549975012,
439
+ "loss": 3.812421875,
440
+ "step": 5500
441
+ },
442
+ {
443
+ "epoch": 0.112,
444
+ "grad_norm": 0.27660876512527466,
445
+ "learning_rate": 0.0005824845107375853,
446
+ "loss": 3.7727542114257813,
447
+ "step": 5600
448
+ },
449
+ {
450
+ "epoch": 0.114,
451
+ "grad_norm": 0.2439262717962265,
452
+ "learning_rate": 0.0005818423472587571,
453
+ "loss": 3.7664410400390627,
454
+ "step": 5700
455
+ },
456
+ {
457
+ "epoch": 0.116,
458
+ "grad_norm": 0.24943126738071442,
459
+ "learning_rate": 0.0005811889900653269,
460
+ "loss": 3.7883651733398436,
461
+ "step": 5800
462
+ },
463
+ {
464
+ "epoch": 0.118,
465
+ "grad_norm": 0.2556191682815552,
466
+ "learning_rate": 0.0005805244651061771,
467
+ "loss": 3.7454019165039063,
468
+ "step": 5900
469
+ },
470
+ {
471
+ "epoch": 0.12,
472
+ "grad_norm": 0.27899155020713806,
473
+ "learning_rate": 0.0005798487987737322,
474
+ "loss": 3.767793884277344,
475
+ "step": 6000
476
+ },
477
+ {
478
+ "epoch": 0.12,
479
+ "eval_accuracy": 0.33610391076885543,
480
+ "eval_loss": 3.7605059146881104,
481
+ "eval_runtime": 5.697,
482
+ "eval_samples_per_second": 340.705,
483
+ "eval_steps_per_second": 21.415,
484
+ "step": 6000
485
  }
486
  ],
487
  "logging_steps": 100,
 
501
  "attributes": {}
502
  }
503
  },
504
+ "total_flos": 1.8813026304e+17,
505
  "train_batch_size": 120,
506
  "trial_name": null,
507
  "trial_params": null