CodeIsAbstract commited on
Commit
bdd5aaa
·
verified ·
1 Parent(s): 81c36c0

Training in progress, step 20000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5163584c22b230a285b507e8fe2e44bfe9c5791489c457ec3f5dae50ef6a9d4e
3
  size 386379
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:73cb875f1568155797c9b13f85384966dd718f46953534c4b08f677ecf9a6a82
3
  size 386379
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ea5c1de12275b49f3c217c8fb0f7b7f68cd81b837676a213fd5cbc5529e41d84
3
  size 1540661735
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:82e61a71536ab895c8bd16825f7157395f1f1e328b8bc9b70f310b0e401ae1c9
3
  size 1540661735
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6d32ac62864aa24d38961f967a353dbdc0c209a83a9a14bd5c6430acf5218f70
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8d96b0f1732638e94f0dd38858b482c7b94b7f3411e8a578ab9a90fdbb757ebb
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:af8916051f16387fb69685f7e072b99274ccf08a163a3cca2253ee5f3a7d6975
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fd4b6599a59298e453c4d085cd48234a781aec741f5259b079ae63537e07c1c7
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:838f0bffb8a21a934a0dd82575abe5438ad86fa4e56f88f4502576bfc1452b15
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c2e7187571674580cade9fd9dfacd95fa924ade19964d58e7bb3223bdcdbc044
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a4277fe562ff3ebc8bf0029b649400f1d520bbe1f14e9fb0ddeb5516160e4f1e
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2cd1fa900a34ad14997cdf150e9de3df071aa3fe4c05fe4db1c61b32b2a4740c
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2aca493fa7cb3a1e93f1a9b8b666fdc3f1dff09789426df3cc65d5126aa02b10
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:82990efadc5b10e2a504af1bdd4a3367edbdc97a4dc60703abf25017a2c875ac
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:454e7b124610176fbeace81b35b4d3891f06294dd70b8c450a31971fb9d0f6e5
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6c9abb1ba3e1b7ce8e555b49390f991a3b537fbfb6d6b26e1bd0c291d5a66e7a
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8a2463c5e875ad508adb42e7029d59e6e8bcbb4d858d11906a1c0881a41dee45
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9d0f333ee4fb3004963191fa2d8aad29bd882472216b178dad7bbc0e452c4074
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8f3ee3875e552d5e8fd915d17b82bcf736028026d7ebd8fb029b832116243bc7
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fc906b60e410824a7dd383e3b2d2e5d2a4a21124b5941e8d66cd1dff093e4f33
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:218c7072bdbb7a0145270cf1a07d80a739b3014f750ae43e7b4b16e30b41a003
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:aab6b99bf19a139d59a98baf0538093fa1ac1d789fb675ce2d1672f97dd7cb0e
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.72,
6
  "eval_steps": 1000,
7
- "global_step": 18000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -1430,6 +1430,164 @@
1430
  "eval_samples_per_second": 17.236,
1431
  "eval_steps_per_second": 0.552,
1432
  "step": 18000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1433
  }
1434
  ],
1435
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.8,
6
  "eval_steps": 1000,
7
+ "global_step": 20000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
1430
  "eval_samples_per_second": 17.236,
1431
  "eval_steps_per_second": 0.552,
1432
  "step": 18000
1433
+ },
1434
+ {
1435
+ "epoch": 0.724,
1436
+ "grad_norm": 1.7778606414794922,
1437
+ "learning_rate": 0.0007066090110867782,
1438
+ "loss": 42.758095703125,
1439
+ "step": 18100
1440
+ },
1441
+ {
1442
+ "epoch": 0.728,
1443
+ "grad_norm": 1.6367878913879395,
1444
+ "learning_rate": 0.0006875340925068554,
1445
+ "loss": 42.4748291015625,
1446
+ "step": 18200
1447
+ },
1448
+ {
1449
+ "epoch": 0.732,
1450
+ "grad_norm": 1.4886348247528076,
1451
+ "learning_rate": 0.0006686665934085281,
1452
+ "loss": 42.22041015625,
1453
+ "step": 18300
1454
+ },
1455
+ {
1456
+ "epoch": 0.736,
1457
+ "grad_norm": 1.0366780757904053,
1458
+ "learning_rate": 0.0006500094955735401,
1459
+ "loss": 42.1452880859375,
1460
+ "step": 18400
1461
+ },
1462
+ {
1463
+ "epoch": 0.74,
1464
+ "grad_norm": 1.189237117767334,
1465
+ "learning_rate": 0.0006315657475322411,
1466
+ "loss": 42.422060546875,
1467
+ "step": 18500
1468
+ },
1469
+ {
1470
+ "epoch": 0.744,
1471
+ "grad_norm": 1.5588538646697998,
1472
+ "learning_rate": 0.0006133382640976061,
1473
+ "loss": 42.59294921875,
1474
+ "step": 18600
1475
+ },
1476
+ {
1477
+ "epoch": 0.748,
1478
+ "grad_norm": 1.2961610555648804,
1479
+ "learning_rate": 0.0005953299259045865,
1480
+ "loss": 42.5683740234375,
1481
+ "step": 18700
1482
+ },
1483
+ {
1484
+ "epoch": 0.752,
1485
+ "grad_norm": 1.513845682144165,
1486
+ "learning_rate": 0.000577543578954858,
1487
+ "loss": 42.31896484375,
1488
+ "step": 18800
1489
+ },
1490
+ {
1491
+ "epoch": 0.756,
1492
+ "grad_norm": 1.0572335720062256,
1493
+ "learning_rate": 0.0005599820341670454,
1494
+ "loss": 42.3233642578125,
1495
+ "step": 18900
1496
+ },
1497
+ {
1498
+ "epoch": 0.76,
1499
+ "grad_norm": 1.4302053451538086,
1500
+ "learning_rate": 0.0005426480669324906,
1501
+ "loss": 42.2752880859375,
1502
+ "step": 19000
1503
+ },
1504
+ {
1505
+ "epoch": 0.76,
1506
+ "eval_accuracy": 0.22522482893450635,
1507
+ "eval_loss": 42.213226318359375,
1508
+ "eval_runtime": 7.0911,
1509
+ "eval_samples_per_second": 17.628,
1510
+ "eval_steps_per_second": 0.564,
1511
+ "step": 19000
1512
+ },
1513
+ {
1514
+ "epoch": 0.764,
1515
+ "grad_norm": 1.4170910120010376,
1516
+ "learning_rate": 0.0005255444166766354,
1517
+ "loss": 42.50333984375,
1518
+ "step": 19100
1519
+ },
1520
+ {
1521
+ "epoch": 0.768,
1522
+ "grad_norm": 1.3892245292663574,
1523
+ "learning_rate": 0.0005086737864260862,
1524
+ "loss": 42.474443359375,
1525
+ "step": 19200
1526
+ },
1527
+ {
1528
+ "epoch": 0.772,
1529
+ "grad_norm": 1.5735739469528198,
1530
+ "learning_rate": 0.0004920388423814364,
1531
+ "loss": 42.25490234375,
1532
+ "step": 19300
1533
+ },
1534
+ {
1535
+ "epoch": 0.776,
1536
+ "grad_norm": 1.2609901428222656,
1537
+ "learning_rate": 0.0004756422134959031,
1538
+ "loss": 42.3439208984375,
1539
+ "step": 19400
1540
+ },
1541
+ {
1542
+ "epoch": 0.78,
1543
+ "grad_norm": 1.1414227485656738,
1544
+ "learning_rate": 0.00045948649105985397,
1545
+ "loss": 42.31787109375,
1546
+ "step": 19500
1547
+ },
1548
+ {
1549
+ "epoch": 0.784,
1550
+ "grad_norm": 1.301095962524414,
1551
+ "learning_rate": 0.0004435742282912836,
1552
+ "loss": 42.43177734375,
1553
+ "step": 19600
1554
+ },
1555
+ {
1556
+ "epoch": 0.788,
1557
+ "grad_norm": 1.1665655374526978,
1558
+ "learning_rate": 0.0004279079399323087,
1559
+ "loss": 42.386767578125,
1560
+ "step": 19700
1561
+ },
1562
+ {
1563
+ "epoch": 0.792,
1564
+ "grad_norm": 1.5682238340377808,
1565
+ "learning_rate": 0.0004124901018517433,
1566
+ "loss": 42.2175439453125,
1567
+ "step": 19800
1568
+ },
1569
+ {
1570
+ "epoch": 0.796,
1571
+ "grad_norm": 1.259811520576477,
1572
+ "learning_rate": 0.00039732315065381754,
1573
+ "loss": 42.252421875,
1574
+ "step": 19900
1575
+ },
1576
+ {
1577
+ "epoch": 0.8,
1578
+ "grad_norm": 0.8927878141403198,
1579
+ "learning_rate": 0.00038240948329310156,
1580
+ "loss": 42.2313232421875,
1581
+ "step": 20000
1582
+ },
1583
+ {
1584
+ "epoch": 0.8,
1585
+ "eval_accuracy": 0.2259481915933529,
1586
+ "eval_loss": 42.14704895019531,
1587
+ "eval_runtime": 6.8564,
1588
+ "eval_samples_per_second": 18.231,
1589
+ "eval_steps_per_second": 0.583,
1590
+ "step": 20000
1591
  }
1592
  ],
1593
  "logging_steps": 100,