CodeIsAbstract commited on
Commit
9714704
·
verified ·
1 Parent(s): 1e61e2e

Training in progress, step 60000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bbc019d20c5bc5141a3e9004bef5a52531dc2a92512d93ad8455f732db0eb48d
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4a3559f22e44a8a648518fcea3f8d97b85f98d79f16bfe2745da3c24e542ca94
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fd70cbbec6d1891a81fb67c9296a4bf20c7702703681faa48b7e22206072facc
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:db7a3532ebb932b79a0341be80d04d64f85c81470da944cf60a0f259b00b3e37
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3fcedeb5e64ebf070876b0c1c4c22aebd04ef9acb651910f4b487df10455e11d
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c4fb849fc4647980403a7edf5a9af1697f444456121347b0f9e1834ac737e420
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:46e948248514973c2ef1b72e4ce8dd6fba8939559c40be1ef19c251df415a7e3
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2a97b2afb49bff977115e649631e564b8b83f6947e290d5e2212d7d11b04765d
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.18181818181818182,
6
  "eval_steps": 1000,
7
- "global_step": 56000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -4376,6 +4376,318 @@
4376
  "eval_samples_per_second": 77.227,
4377
  "eval_steps_per_second": 19.307,
4378
  "step": 56000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4379
  }
4380
  ],
4381
  "logging_steps": 100,
@@ -4395,7 +4707,7 @@
4395
  "attributes": {}
4396
  }
4397
  },
4398
- "total_flos": 1.395216535781376e+18,
4399
  "train_batch_size": 22,
4400
  "trial_name": null,
4401
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.21818181818181817,
6
  "eval_steps": 1000,
7
+ "global_step": 60000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
4376
  "eval_samples_per_second": 77.227,
4377
  "eval_steps_per_second": 19.307,
4378
  "step": 56000
4379
+ },
4380
+ {
4381
+ "epoch": 0.18272727272727274,
4382
+ "grad_norm": 0.15831124782562256,
4383
+ "learning_rate": 0.00048535943984463943,
4384
+ "loss": 2.685963134765625,
4385
+ "step": 56100
4386
+ },
4387
+ {
4388
+ "epoch": 0.18363636363636363,
4389
+ "grad_norm": 0.17237283289432526,
4390
+ "learning_rate": 0.00048393016821573625,
4391
+ "loss": 2.652848205566406,
4392
+ "step": 56200
4393
+ },
4394
+ {
4395
+ "epoch": 0.18454545454545454,
4396
+ "grad_norm": 0.172745019197464,
4397
+ "learning_rate": 0.00048250102802172067,
4398
+ "loss": 2.6385073852539063,
4399
+ "step": 56300
4400
+ },
4401
+ {
4402
+ "epoch": 0.18545454545454546,
4403
+ "grad_norm": 0.16651754081249237,
4404
+ "learning_rate": 0.00048107203095150636,
4405
+ "loss": 2.6532525634765625,
4406
+ "step": 56400
4407
+ },
4408
+ {
4409
+ "epoch": 0.18636363636363637,
4410
+ "grad_norm": 0.16987019777297974,
4411
+ "learning_rate": 0.00047964318869283683,
4412
+ "loss": 2.690726623535156,
4413
+ "step": 56500
4414
+ },
4415
+ {
4416
+ "epoch": 0.18727272727272729,
4417
+ "grad_norm": 0.15424904227256775,
4418
+ "learning_rate": 0.000478214512932189,
4419
+ "loss": 2.6622409057617187,
4420
+ "step": 56600
4421
+ },
4422
+ {
4423
+ "epoch": 0.18818181818181817,
4424
+ "grad_norm": 0.1839853674173355,
4425
+ "learning_rate": 0.0004767860153546783,
4426
+ "loss": 2.683760681152344,
4427
+ "step": 56700
4428
+ },
4429
+ {
4430
+ "epoch": 0.1890909090909091,
4431
+ "grad_norm": 0.15451455116271973,
4432
+ "learning_rate": 0.0004753577076439628,
4433
+ "loss": 2.6614370727539063,
4434
+ "step": 56800
4435
+ },
4436
+ {
4437
+ "epoch": 0.19,
4438
+ "grad_norm": 0.19505156576633453,
4439
+ "learning_rate": 0.0004739296014821474,
4440
+ "loss": 2.65382080078125,
4441
+ "step": 56900
4442
+ },
4443
+ {
4444
+ "epoch": 0.19090909090909092,
4445
+ "grad_norm": 0.20561884343624115,
4446
+ "learning_rate": 0.0004725017085496889,
4447
+ "loss": 2.70877685546875,
4448
+ "step": 57000
4449
+ },
4450
+ {
4451
+ "epoch": 0.19090909090909092,
4452
+ "eval_loss": 3.0625686645507812,
4453
+ "eval_runtime": 7.4104,
4454
+ "eval_samples_per_second": 77.728,
4455
+ "eval_steps_per_second": 19.432,
4456
+ "step": 57000
4457
+ },
4458
+ {
4459
+ "epoch": 0.1918181818181818,
4460
+ "grad_norm": 0.1769583523273468,
4461
+ "learning_rate": 0.00047107404052529957,
4462
+ "loss": 2.6674728393554688,
4463
+ "step": 57100
4464
+ },
4465
+ {
4466
+ "epoch": 0.19272727272727272,
4467
+ "grad_norm": 0.15023104846477509,
4468
+ "learning_rate": 0.0004696466090858528,
4469
+ "loss": 2.6564959716796874,
4470
+ "step": 57200
4471
+ },
4472
+ {
4473
+ "epoch": 0.19363636363636363,
4474
+ "grad_norm": 0.1738213300704956,
4475
+ "learning_rate": 0.00046821942590628645,
4476
+ "loss": 2.6518304443359373,
4477
+ "step": 57300
4478
+ },
4479
+ {
4480
+ "epoch": 0.19454545454545455,
4481
+ "grad_norm": 0.15652106702327728,
4482
+ "learning_rate": 0.00046679250265950815,
4483
+ "loss": 2.6429074096679686,
4484
+ "step": 57400
4485
+ },
4486
+ {
4487
+ "epoch": 0.19545454545454546,
4488
+ "grad_norm": 0.20546045899391174,
4489
+ "learning_rate": 0.00046536585101629945,
4490
+ "loss": 2.66235107421875,
4491
+ "step": 57500
4492
+ },
4493
+ {
4494
+ "epoch": 0.19636363636363635,
4495
+ "grad_norm": 0.15202665328979492,
4496
+ "learning_rate": 0.0004639394826452204,
4497
+ "loss": 2.635772399902344,
4498
+ "step": 57600
4499
+ },
4500
+ {
4501
+ "epoch": 0.19727272727272727,
4502
+ "grad_norm": 0.1581605225801468,
4503
+ "learning_rate": 0.00046251340921251436,
4504
+ "loss": 2.6710745239257814,
4505
+ "step": 57700
4506
+ },
4507
+ {
4508
+ "epoch": 0.19818181818181818,
4509
+ "grad_norm": 0.19863693416118622,
4510
+ "learning_rate": 0.000461087642382012,
4511
+ "loss": 2.645235595703125,
4512
+ "step": 57800
4513
+ },
4514
+ {
4515
+ "epoch": 0.1990909090909091,
4516
+ "grad_norm": 0.14989744126796722,
4517
+ "learning_rate": 0.00045966219381503694,
4518
+ "loss": 2.6374029541015624,
4519
+ "step": 57900
4520
+ },
4521
+ {
4522
+ "epoch": 0.2,
4523
+ "grad_norm": 0.18618331849575043,
4524
+ "learning_rate": 0.000458237075170309,
4525
+ "loss": 2.6277200317382814,
4526
+ "step": 58000
4527
+ },
4528
+ {
4529
+ "epoch": 0.2,
4530
+ "eval_loss": 3.0661063194274902,
4531
+ "eval_runtime": 7.4551,
4532
+ "eval_samples_per_second": 77.262,
4533
+ "eval_steps_per_second": 19.316,
4534
+ "step": 58000
4535
+ },
4536
+ {
4537
+ "epoch": 0.2009090909090909,
4538
+ "grad_norm": 0.16106659173965454,
4539
+ "learning_rate": 0.00045681229810385005,
4540
+ "loss": 2.6352557373046874,
4541
+ "step": 58100
4542
+ },
4543
+ {
4544
+ "epoch": 0.2018181818181818,
4545
+ "grad_norm": 0.14770276844501495,
4546
+ "learning_rate": 0.000455387874268888,
4547
+ "loss": 2.627791442871094,
4548
+ "step": 58200
4549
+ },
4550
+ {
4551
+ "epoch": 0.20272727272727273,
4552
+ "grad_norm": 0.17903557419776917,
4553
+ "learning_rate": 0.0004539638153157621,
4554
+ "loss": 2.638799133300781,
4555
+ "step": 58300
4556
+ },
4557
+ {
4558
+ "epoch": 0.20363636363636364,
4559
+ "grad_norm": 0.16185429692268372,
4560
+ "learning_rate": 0.0004525401328918264,
4561
+ "loss": 2.635686340332031,
4562
+ "step": 58400
4563
+ },
4564
+ {
4565
+ "epoch": 0.20454545454545456,
4566
+ "grad_norm": 0.17657916247844696,
4567
+ "learning_rate": 0.00045111683864135617,
4568
+ "loss": 2.6360543823242186,
4569
+ "step": 58500
4570
+ },
4571
+ {
4572
+ "epoch": 0.20545454545454545,
4573
+ "grad_norm": 0.14804042875766754,
4574
+ "learning_rate": 0.0004496939442054514,
4575
+ "loss": 2.643255310058594,
4576
+ "step": 58600
4577
+ },
4578
+ {
4579
+ "epoch": 0.20636363636363636,
4580
+ "grad_norm": 0.16301922500133514,
4581
+ "learning_rate": 0.0004482714612219419,
4582
+ "loss": 2.64661376953125,
4583
+ "step": 58700
4584
+ },
4585
+ {
4586
+ "epoch": 0.20727272727272728,
4587
+ "grad_norm": 0.1726071685552597,
4588
+ "learning_rate": 0.00044684940132529264,
4589
+ "loss": 2.6665936279296876,
4590
+ "step": 58800
4591
+ },
4592
+ {
4593
+ "epoch": 0.2081818181818182,
4594
+ "grad_norm": 0.16906021535396576,
4595
+ "learning_rate": 0.0004454277761465076,
4596
+ "loss": 2.65294921875,
4597
+ "step": 58900
4598
+ },
4599
+ {
4600
+ "epoch": 0.20909090909090908,
4601
+ "grad_norm": 0.17039760947227478,
4602
+ "learning_rate": 0.0004440065973130358,
4603
+ "loss": 2.6539688110351562,
4604
+ "step": 59000
4605
+ },
4606
+ {
4607
+ "epoch": 0.20909090909090908,
4608
+ "eval_loss": 3.057651996612549,
4609
+ "eval_runtime": 7.4341,
4610
+ "eval_samples_per_second": 77.481,
4611
+ "eval_steps_per_second": 19.37,
4612
+ "step": 59000
4613
+ },
4614
+ {
4615
+ "epoch": 0.21,
4616
+ "grad_norm": 0.1761595457792282,
4617
+ "learning_rate": 0.0004425858764486751,
4618
+ "loss": 2.6766180419921874,
4619
+ "step": 59100
4620
+ },
4621
+ {
4622
+ "epoch": 0.2109090909090909,
4623
+ "grad_norm": 0.16640876233577728,
4624
+ "learning_rate": 0.00044116562517347796,
4625
+ "loss": 2.6426458740234375,
4626
+ "step": 59200
4627
+ },
4628
+ {
4629
+ "epoch": 0.21181818181818182,
4630
+ "grad_norm": 0.1541028618812561,
4631
+ "learning_rate": 0.00043974585510365614,
4632
+ "loss": 2.6716134643554685,
4633
+ "step": 59300
4634
+ },
4635
+ {
4636
+ "epoch": 0.21272727272727274,
4637
+ "grad_norm": 0.16134560108184814,
4638
+ "learning_rate": 0.0004383265778514852,
4639
+ "loss": 2.6808517456054686,
4640
+ "step": 59400
4641
+ },
4642
+ {
4643
+ "epoch": 0.21363636363636362,
4644
+ "grad_norm": 0.1456531435251236,
4645
+ "learning_rate": 0.0004369078050252105,
4646
+ "loss": 2.684778747558594,
4647
+ "step": 59500
4648
+ },
4649
+ {
4650
+ "epoch": 0.21454545454545454,
4651
+ "grad_norm": 0.1792006641626358,
4652
+ "learning_rate": 0.00043548954822895117,
4653
+ "loss": 2.649779052734375,
4654
+ "step": 59600
4655
+ },
4656
+ {
4657
+ "epoch": 0.21545454545454545,
4658
+ "grad_norm": 0.18216730654239655,
4659
+ "learning_rate": 0.00043407181906260627,
4660
+ "loss": 2.677138671875,
4661
+ "step": 59700
4662
+ },
4663
+ {
4664
+ "epoch": 0.21636363636363637,
4665
+ "grad_norm": 0.1598980575799942,
4666
+ "learning_rate": 0.0004326546291217589,
4667
+ "loss": 2.6525616455078125,
4668
+ "step": 59800
4669
+ },
4670
+ {
4671
+ "epoch": 0.21727272727272728,
4672
+ "grad_norm": 0.16436710953712463,
4673
+ "learning_rate": 0.0004312379899975821,
4674
+ "loss": 2.6850616455078127,
4675
+ "step": 59900
4676
+ },
4677
+ {
4678
+ "epoch": 0.21818181818181817,
4679
+ "grad_norm": 0.175828754901886,
4680
+ "learning_rate": 0.00042982191327674404,
4681
+ "loss": 2.6756378173828126,
4682
+ "step": 60000
4683
+ },
4684
+ {
4685
+ "epoch": 0.21818181818181817,
4686
+ "eval_loss": 3.0540809631347656,
4687
+ "eval_runtime": 7.4118,
4688
+ "eval_samples_per_second": 77.714,
4689
+ "eval_steps_per_second": 19.429,
4690
+ "step": 60000
4691
  }
4692
  ],
4693
  "logging_steps": 100,
 
4707
  "attributes": {}
4708
  }
4709
  },
4710
+ "total_flos": 1.49487485976576e+18,
4711
  "train_batch_size": 22,
4712
  "trial_name": null,
4713
  "trial_params": null