CodeIsAbstract commited on
Commit
1cda29f
·
verified ·
1 Parent(s): f7df0c9

Training in progress, step 20000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:91f77bd1ac2f3fafb2d2dff8abe3cf8b6a15fc5e2c266220cfd41101a08926f2
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0ba3c2b02c2181522825a50de924743eb11a5c8ec14f96218442ef3bf9c669ad
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9bd91aa34b703c596d190d43204798100682f3f8c6905d18b1c963356249d128
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:683d6ef85df4768d6187df8f33dd7d019dd22439aeb22f78e278b0cdd0d17a92
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2118ba9622183c33aa1d1262712b3091d1f33a50e27c8636e086286e6c061b97
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5451e921407cd3b87f821062b7ea647b65f6d0ce5a09f45a2e5fe63f79c5afdc
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:803e8ad8cb04d8d95ab5fe814114135859d42a7ef5d29f0ea36f7faef8803122
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fdc6d60c7c4fe4f9e140eb1a01d69a125cb28ae84f348f2ce7bfdb028834e194
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.36,
6
  "eval_steps": 1000,
7
- "global_step": 18000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1430,6 +1430,164 @@
1430
  "eval_samples_per_second": 287.759,
1431
  "eval_steps_per_second": 18.087,
1432
  "step": 18000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1433
  }
1434
  ],
1435
  "logging_steps": 100,
@@ -1449,7 +1607,7 @@
1449
  "attributes": {}
1450
  }
1451
  },
1452
- "total_flos": 7.0559295602688e+17,
1453
  "train_batch_size": 120,
1454
  "trial_name": null,
1455
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.4,
6
  "eval_steps": 1000,
7
+ "global_step": 20000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1430
  "eval_samples_per_second": 287.759,
1431
  "eval_steps_per_second": 18.087,
1432
  "step": 18000
1433
+ },
1434
+ {
1435
+ "epoch": 0.362,
1436
+ "grad_norm": 0.20023606717586517,
1437
+ "learning_rate": 0.0007128041074339486,
1438
+ "loss": 3.4777850341796874,
1439
+ "step": 18100
1440
+ },
1441
+ {
1442
+ "epoch": 0.364,
1443
+ "grad_norm": 0.4378589689731598,
1444
+ "learning_rate": 0.0007099484953400711,
1445
+ "loss": 3.483775634765625,
1446
+ "step": 18200
1447
+ },
1448
+ {
1449
+ "epoch": 0.366,
1450
+ "grad_norm": 0.15016524493694305,
1451
+ "learning_rate": 0.0007070845448841241,
1452
+ "loss": 3.471934509277344,
1453
+ "step": 18300
1454
+ },
1455
+ {
1456
+ "epoch": 0.368,
1457
+ "grad_norm": 0.17178508639335632,
1458
+ "learning_rate": 0.0007042123698114138,
1459
+ "loss": 3.465678405761719,
1460
+ "step": 18400
1461
+ },
1462
+ {
1463
+ "epoch": 0.37,
1464
+ "grad_norm": 0.17568770051002502,
1465
+ "learning_rate": 0.0007013320841938966,
1466
+ "loss": 3.484932556152344,
1467
+ "step": 18500
1468
+ },
1469
+ {
1470
+ "epoch": 0.372,
1471
+ "grad_norm": 0.12633204460144043,
1472
+ "learning_rate": 0.0006984438024256499,
1473
+ "loss": 3.4721084594726563,
1474
+ "step": 18600
1475
+ },
1476
+ {
1477
+ "epoch": 0.374,
1478
+ "grad_norm": 0.13481932878494263,
1479
+ "learning_rate": 0.0006955476392183275,
1480
+ "loss": 3.441580505371094,
1481
+ "step": 18700
1482
+ },
1483
+ {
1484
+ "epoch": 0.376,
1485
+ "grad_norm": 0.142787367105484,
1486
+ "learning_rate": 0.0006926437095966044,
1487
+ "loss": 3.472591552734375,
1488
+ "step": 18800
1489
+ },
1490
+ {
1491
+ "epoch": 0.378,
1492
+ "grad_norm": 0.15913543105125427,
1493
+ "learning_rate": 0.000689732128893608,
1494
+ "loss": 3.4664480590820315,
1495
+ "step": 18900
1496
+ },
1497
+ {
1498
+ "epoch": 0.38,
1499
+ "grad_norm": 0.5166953802108765,
1500
+ "learning_rate": 0.0006868130127463385,
1501
+ "loss": 3.44828125,
1502
+ "step": 19000
1503
+ },
1504
+ {
1505
+ "epoch": 0.38,
1506
+ "eval_accuracy": 0.36941738224793846,
1507
+ "eval_loss": 3.4189138412475586,
1508
+ "eval_runtime": 6.3949,
1509
+ "eval_samples_per_second": 303.523,
1510
+ "eval_steps_per_second": 19.078,
1511
+ "step": 19000
1512
+ },
1513
+ {
1514
+ "epoch": 0.382,
1515
+ "grad_norm": 0.138035848736763,
1516
+ "learning_rate": 0.0006838864770910745,
1517
+ "loss": 3.4636468505859375,
1518
+ "step": 19100
1519
+ },
1520
+ {
1521
+ "epoch": 0.384,
1522
+ "grad_norm": 0.2230982631444931,
1523
+ "learning_rate": 0.0006809526381587703,
1524
+ "loss": 3.4643569946289063,
1525
+ "step": 19200
1526
+ },
1527
+ {
1528
+ "epoch": 0.386,
1529
+ "grad_norm": 0.12826211750507355,
1530
+ "learning_rate": 0.0006780116124704383,
1531
+ "loss": 3.45958984375,
1532
+ "step": 19300
1533
+ },
1534
+ {
1535
+ "epoch": 0.388,
1536
+ "grad_norm": 0.1421106457710266,
1537
+ "learning_rate": 0.0006750635168325223,
1538
+ "loss": 3.4553872680664064,
1539
+ "step": 19400
1540
+ },
1541
+ {
1542
+ "epoch": 0.39,
1543
+ "grad_norm": 0.14258891344070435,
1544
+ "learning_rate": 0.000672108468332257,
1545
+ "loss": 3.4486114501953127,
1546
+ "step": 19500
1547
+ },
1548
+ {
1549
+ "epoch": 0.392,
1550
+ "grad_norm": 0.24561843276023865,
1551
+ "learning_rate": 0.0006691465843330194,
1552
+ "loss": 3.4674017333984377,
1553
+ "step": 19600
1554
+ },
1555
+ {
1556
+ "epoch": 0.394,
1557
+ "grad_norm": 0.13396546244621277,
1558
+ "learning_rate": 0.0006661779824696658,
1559
+ "loss": 3.4596041870117187,
1560
+ "step": 19700
1561
+ },
1562
+ {
1563
+ "epoch": 0.396,
1564
+ "grad_norm": 0.145228311419487,
1565
+ "learning_rate": 0.000663202780643862,
1566
+ "loss": 3.450461120605469,
1567
+ "step": 19800
1568
+ },
1569
+ {
1570
+ "epoch": 0.398,
1571
+ "grad_norm": 0.1595899611711502,
1572
+ "learning_rate": 0.0006602210970193982,
1573
+ "loss": 3.4370431518554687,
1574
+ "step": 19900
1575
+ },
1576
+ {
1577
+ "epoch": 0.4,
1578
+ "grad_norm": 0.1425502896308899,
1579
+ "learning_rate": 0.0006572330500174975,
1580
+ "loss": 3.441788330078125,
1581
+ "step": 20000
1582
+ },
1583
+ {
1584
+ "epoch": 0.4,
1585
+ "eval_accuracy": 0.3703691381064293,
1586
+ "eval_loss": 3.4063475131988525,
1587
+ "eval_runtime": 6.6969,
1588
+ "eval_samples_per_second": 289.838,
1589
+ "eval_steps_per_second": 18.218,
1590
+ "step": 20000
1591
  }
1592
  ],
1593
  "logging_steps": 100,
 
1607
  "attributes": {}
1608
  }
1609
  },
1610
+ "total_flos": 7.839921733632e+17,
1611
  "train_batch_size": 120,
1612
  "trial_name": null,
1613
  "trial_params": null