CodeIsAbstract commited on
Commit
56366b2
·
verified ·
1 Parent(s): 7778718

Training in progress, step 2100, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e1fdfb8dfcb901505d88cb8a470e308efd9e0925381c391ffc384136b1ebd88d
3
  size 847599616
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c23197ba0795edc47fdff663e5bff5ca6f2da2dbd852fe61bc5dad03f91766ca
3
  size 847599616
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fec21f808a923bf61ae943ab35e94b8cc5c8bae8a15db1f8728417fdb5514516
3
  size 1386414411
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5b5536f030012f5595a66c66ab0a774aca6ec0b33b4068366d43f44299a8e261
3
  size 1386414411
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:29ddf99768976dfa1ee6d0925d35e9cc22f598f3ea5118aa848b2b6342be72c1
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6ad904a911703987c832a328a0d981e81238ebdc07028f410b30c9d17ed5504d
3
  size 14917
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5fd13efabdb0e66816b180aa0d21e252da318cce2ddf5d9d38e39eb7b38ab940
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8da0cb13086084a45a6c36c3946eff5411f17ac0d264fc6d429bffa46fef2d07
3
  size 14917
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:423108cec97a3ca65f7a856a6a60f5138d6147e68a603ecfb19ab83adf2f9fe9
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cbf265287bee4c56f362ef367ee6be0abeb4ae512cb586931cce160cfc0287c4
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.8571428571428571,
6
  "eval_steps": 150,
7
- "global_step": 1800,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1376,6 +1376,234 @@
1376
  "eval_samples_per_second": 19.703,
1377
  "eval_steps_per_second": 2.483,
1378
  "step": 1800
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1379
  }
1380
  ],
1381
  "logging_steps": 10,
@@ -1390,7 +1618,7 @@
1390
  "should_evaluate": false,
1391
  "should_log": false,
1392
  "should_save": true,
1393
- "should_training_stop": false
1394
  },
1395
  "attributes": {}
1396
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 1.0,
6
  "eval_steps": 150,
7
+ "global_step": 2100,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1376
  "eval_samples_per_second": 19.703,
1377
  "eval_steps_per_second": 2.483,
1378
  "step": 1800
1379
+ },
1380
+ {
1381
+ "epoch": 0.861904761904762,
1382
+ "grad_norm": 15.835714340209961,
1383
+ "learning_rate": 2.0634158645551358e-07,
1384
+ "loss": 16.231326293945312,
1385
+ "step": 1810
1386
+ },
1387
+ {
1388
+ "epoch": 0.8666666666666667,
1389
+ "grad_norm": 5.838013648986816,
1390
+ "learning_rate": 1.9263203866925725e-07,
1391
+ "loss": 16.3001220703125,
1392
+ "step": 1820
1393
+ },
1394
+ {
1395
+ "epoch": 0.8714285714285714,
1396
+ "grad_norm": 8.052929878234863,
1397
+ "learning_rate": 1.7937066993133553e-07,
1398
+ "loss": 16.139373779296875,
1399
+ "step": 1830
1400
+ },
1401
+ {
1402
+ "epoch": 0.8761904761904762,
1403
+ "grad_norm": 18.70516586303711,
1404
+ "learning_rate": 1.665607687074886e-07,
1405
+ "loss": 16.020474243164063,
1406
+ "step": 1840
1407
+ },
1408
+ {
1409
+ "epoch": 0.8809523809523809,
1410
+ "grad_norm": 17.08431053161621,
1411
+ "learning_rate": 1.5420551151155393e-07,
1412
+ "loss": 16.245944213867187,
1413
+ "step": 1850
1414
+ },
1415
+ {
1416
+ "epoch": 0.8857142857142857,
1417
+ "grad_norm": 17.282939910888672,
1418
+ "learning_rate": 1.4230796211777784e-07,
1419
+ "loss": 16.05218505859375,
1420
+ "step": 1860
1421
+ },
1422
+ {
1423
+ "epoch": 0.8904761904761904,
1424
+ "grad_norm": 13.214439392089844,
1425
+ "learning_rate": 1.3087107080107852e-07,
1426
+ "loss": 16.017184448242187,
1427
+ "step": 1870
1428
+ },
1429
+ {
1430
+ "epoch": 0.8952380952380953,
1431
+ "grad_norm": 22.98546600341797,
1432
+ "learning_rate": 1.1989767360545779e-07,
1433
+ "loss": 16.036334228515624,
1434
+ "step": 1880
1435
+ },
1436
+ {
1437
+ "epoch": 0.9,
1438
+ "grad_norm": 23.661928176879883,
1439
+ "learning_rate": 1.0939049164073777e-07,
1440
+ "loss": 15.678387451171876,
1441
+ "step": 1890
1442
+ },
1443
+ {
1444
+ "epoch": 0.9047619047619048,
1445
+ "grad_norm": 11.723611831665039,
1446
+ "learning_rate": 9.935213040779911e-08,
1447
+ "loss": 16.37012939453125,
1448
+ "step": 1900
1449
+ },
1450
+ {
1451
+ "epoch": 0.9095238095238095,
1452
+ "grad_norm": 21.208452224731445,
1453
+ "learning_rate": 8.978507915248434e-08,
1454
+ "loss": 16.12847900390625,
1455
+ "step": 1910
1456
+ },
1457
+ {
1458
+ "epoch": 0.9142857142857143,
1459
+ "grad_norm": 19.140209197998047,
1460
+ "learning_rate": 8.069171024833222e-08,
1461
+ "loss": 16.41150817871094,
1462
+ "step": 1920
1463
+ },
1464
+ {
1465
+ "epoch": 0.919047619047619,
1466
+ "grad_norm": 12.952860832214355,
1467
+ "learning_rate": 7.207427860829351e-08,
1468
+ "loss": 15.9720703125,
1469
+ "step": 1930
1470
+ },
1471
+ {
1472
+ "epoch": 0.9238095238095239,
1473
+ "grad_norm": 11.641100883483887,
1474
+ "learning_rate": 6.393492112557064e-08,
1475
+ "loss": 16.332427978515625,
1476
+ "step": 1940
1477
+ },
1478
+ {
1479
+ "epoch": 0.9285714285714286,
1480
+ "grad_norm": 10.129194259643555,
1481
+ "learning_rate": 5.627565614372653e-08,
1482
+ "loss": 16.232325744628906,
1483
+ "step": 1950
1484
+ },
1485
+ {
1486
+ "epoch": 0.9285714285714286,
1487
+ "eval_accuracy": 0.06559055118110237,
1488
+ "eval_loss": 16.09013557434082,
1489
+ "eval_runtime": 25.2262,
1490
+ "eval_samples_per_second": 19.821,
1491
+ "eval_steps_per_second": 2.497,
1492
+ "step": 1950
1493
+ },
1494
+ {
1495
+ "epoch": 0.9333333333333333,
1496
+ "grad_norm": 10.955820083618164,
1497
+ "learning_rate": 4.909838295618907e-08,
1498
+ "loss": 16.038211059570312,
1499
+ "step": 1960
1500
+ },
1501
+ {
1502
+ "epoch": 0.9380952380952381,
1503
+ "grad_norm": 10.441788673400879,
1504
+ "learning_rate": 4.2404881335276644e-08,
1505
+ "loss": 16.02879638671875,
1506
+ "step": 1970
1507
+ },
1508
+ {
1509
+ "epoch": 0.9428571428571428,
1510
+ "grad_norm": 9.26267147064209,
1511
+ "learning_rate": 3.619681109086237e-08,
1512
+ "loss": 15.950970458984376,
1513
+ "step": 1980
1514
+ },
1515
+ {
1516
+ "epoch": 0.9476190476190476,
1517
+ "grad_norm": 19.122802734375,
1518
+ "learning_rate": 3.0475711658785485e-08,
1519
+ "loss": 16.08349151611328,
1520
+ "step": 1990
1521
+ },
1522
+ {
1523
+ "epoch": 0.9523809523809523,
1524
+ "grad_norm": 18.312496185302734,
1525
+ "learning_rate": 2.5243001719111646e-08,
1526
+ "loss": 16.050738525390624,
1527
+ "step": 2000
1528
+ },
1529
+ {
1530
+ "epoch": 0.9571428571428572,
1531
+ "grad_norm": 11.455791473388672,
1532
+ "learning_rate": 2.049997884433985e-08,
1533
+ "loss": 16.05840606689453,
1534
+ "step": 2010
1535
+ },
1536
+ {
1537
+ "epoch": 0.9619047619047619,
1538
+ "grad_norm": 11.550554275512695,
1539
+ "learning_rate": 1.6247819177636735e-08,
1540
+ "loss": 16.087054443359374,
1541
+ "step": 2020
1542
+ },
1543
+ {
1544
+ "epoch": 0.9666666666666667,
1545
+ "grad_norm": 19.676589965820312,
1546
+ "learning_rate": 1.248757714118609e-08,
1547
+ "loss": 15.8768310546875,
1548
+ "step": 2030
1549
+ },
1550
+ {
1551
+ "epoch": 0.9714285714285714,
1552
+ "grad_norm": 9.593478202819824,
1553
+ "learning_rate": 9.220185174720453e-09,
1554
+ "loss": 16.02369079589844,
1555
+ "step": 2040
1556
+ },
1557
+ {
1558
+ "epoch": 0.9761904761904762,
1559
+ "grad_norm": 10.647096633911133,
1560
+ "learning_rate": 6.446453504299176e-09,
1561
+ "loss": 15.94654541015625,
1562
+ "step": 2050
1563
+ },
1564
+ {
1565
+ "epoch": 0.9809523809523809,
1566
+ "grad_norm": 6.6210713386535645,
1567
+ "learning_rate": 4.167069941395818e-09,
1568
+ "loss": 16.178671264648436,
1569
+ "step": 2060
1570
+ },
1571
+ {
1572
+ "epoch": 0.9857142857142858,
1573
+ "grad_norm": 11.91507339477539,
1574
+ "learning_rate": 2.38259971233834e-09,
1575
+ "loss": 15.88709716796875,
1576
+ "step": 2070
1577
+ },
1578
+ {
1579
+ "epoch": 0.9904761904761905,
1580
+ "grad_norm": 29.012462615966797,
1581
+ "learning_rate": 1.093485318147902e-09,
1582
+ "loss": 15.998876953125,
1583
+ "step": 2080
1584
+ },
1585
+ {
1586
+ "epoch": 0.9952380952380953,
1587
+ "grad_norm": 13.495532035827637,
1588
+ "learning_rate": 3.0004642481151755e-10,
1589
+ "loss": 15.837295532226562,
1590
+ "step": 2090
1591
+ },
1592
+ {
1593
+ "epoch": 1.0,
1594
+ "grad_norm": 10.239914894104004,
1595
+ "learning_rate": 2.4797840116885795e-12,
1596
+ "loss": 16.02972869873047,
1597
+ "step": 2100
1598
+ },
1599
+ {
1600
+ "epoch": 1.0,
1601
+ "eval_accuracy": 0.06144094488188977,
1602
+ "eval_loss": 16.065017700195312,
1603
+ "eval_runtime": 25.407,
1604
+ "eval_samples_per_second": 19.68,
1605
+ "eval_steps_per_second": 2.48,
1606
+ "step": 2100
1607
  }
1608
  ],
1609
  "logging_steps": 10,
 
1618
  "should_evaluate": false,
1619
  "should_log": false,
1620
  "should_save": true,
1621
+ "should_training_stop": true
1622
  },
1623
  "attributes": {}
1624
  }