CodeIsAbstract commited on
Commit
5cdf9bd
·
verified ·
1 Parent(s): fb1ac75

Training in progress, step 20000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d69e130591023d3408c94e3445de9f9023a881b14d399648ae4226ee5bff553b
3
  size 496262784
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:80f6cba2576f1ee2628aa99fb4980b692c7f23dc4a1cc36b06da1ee070ed4750
3
  size 496262784
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a86f5fd274e719671c5f2463396daaf1a7a1c0ce3b5b26f63b7d962eb3bc34f1
3
  size 992621963
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:238c57d596e4b481c83fa003f41489533d0e0d6fa65bc29cfff61620056f3f43
3
  size 992621963
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7a721edff9706126df51d3abc78a93078e6b8e75513d77f55a62d72e4c5b16cf
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f2d3bdd032f891e90f3fae19dbb6eb0bfaf6ef591769711ba856b31e40586a5e
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5637b5a3d72a0623ba9410834e3d0160dfe8e69a524a3dc9973a65396f79b127
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c5b7f9f4f6c99c7a7ebbb7780b490f0a9830b3a79c1c5a07331e7e6edccccd28
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.36,
6
  "eval_steps": 1000,
7
- "global_step": 18000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1430,6 +1430,164 @@
1430
  "eval_samples_per_second": 306.051,
1431
  "eval_steps_per_second": 19.237,
1432
  "step": 18000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1433
  }
1434
  ],
1435
  "logging_steps": 100,
@@ -1449,7 +1607,7 @@
1449
  "attributes": {}
1450
  }
1451
  },
1452
- "total_flos": 5.6439078912e+17,
1453
  "train_batch_size": 120,
1454
  "trial_name": null,
1455
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.4,
6
  "eval_steps": 1000,
7
+ "global_step": 20000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1430
  "eval_samples_per_second": 306.051,
1431
  "eval_steps_per_second": 19.237,
1432
  "step": 18000
1433
+ },
1434
+ {
1435
+ "epoch": 0.362,
1436
+ "grad_norm": 0.2838000953197479,
1437
+ "learning_rate": 0.0004276824644603691,
1438
+ "loss": 3.5090277099609377,
1439
+ "step": 18100
1440
+ },
1441
+ {
1442
+ "epoch": 0.364,
1443
+ "grad_norm": 0.6395978331565857,
1444
+ "learning_rate": 0.0004259690972040426,
1445
+ "loss": 3.5145654296875,
1446
+ "step": 18200
1447
+ },
1448
+ {
1449
+ "epoch": 0.366,
1450
+ "grad_norm": 0.1982821524143219,
1451
+ "learning_rate": 0.0004242507269304744,
1452
+ "loss": 3.504261474609375,
1453
+ "step": 18300
1454
+ },
1455
+ {
1456
+ "epoch": 0.368,
1457
+ "grad_norm": 0.22082549333572388,
1458
+ "learning_rate": 0.0004225274218868482,
1459
+ "loss": 3.496566162109375,
1460
+ "step": 18400
1461
+ },
1462
+ {
1463
+ "epoch": 0.37,
1464
+ "grad_norm": 0.2385374903678894,
1465
+ "learning_rate": 0.0004207992505163379,
1466
+ "loss": 3.514676208496094,
1467
+ "step": 18500
1468
+ },
1469
+ {
1470
+ "epoch": 0.372,
1471
+ "grad_norm": 0.18938113749027252,
1472
+ "learning_rate": 0.00041906628145538987,
1473
+ "loss": 3.5016729736328127,
1474
+ "step": 18600
1475
+ },
1476
+ {
1477
+ "epoch": 0.374,
1478
+ "grad_norm": 0.17800572514533997,
1479
+ "learning_rate": 0.0004173285835309965,
1480
+ "loss": 3.471422119140625,
1481
+ "step": 18700
1482
+ },
1483
+ {
1484
+ "epoch": 0.376,
1485
+ "grad_norm": 0.19377586245536804,
1486
+ "learning_rate": 0.0004155862257579626,
1487
+ "loss": 3.5025115966796876,
1488
+ "step": 18800
1489
+ },
1490
+ {
1491
+ "epoch": 0.378,
1492
+ "grad_norm": 0.21612316370010376,
1493
+ "learning_rate": 0.00041383927733616477,
1494
+ "loss": 3.4965274047851564,
1495
+ "step": 18900
1496
+ },
1497
+ {
1498
+ "epoch": 0.38,
1499
+ "grad_norm": 0.6052186489105225,
1500
+ "learning_rate": 0.000412087807647803,
1501
+ "loss": 3.4777236938476563,
1502
+ "step": 19000
1503
+ },
1504
+ {
1505
+ "epoch": 0.38,
1506
+ "eval_accuracy": 0.3683073364850164,
1507
+ "eval_loss": 3.4274120330810547,
1508
+ "eval_runtime": 6.5604,
1509
+ "eval_samples_per_second": 295.864,
1510
+ "eval_steps_per_second": 18.596,
1511
+ "step": 19000
1512
+ },
1513
+ {
1514
+ "epoch": 0.382,
1515
+ "grad_norm": 0.19751021265983582,
1516
+ "learning_rate": 0.0004103318862546447,
1517
+ "loss": 3.4927557373046874,
1518
+ "step": 19100
1519
+ },
1520
+ {
1521
+ "epoch": 0.384,
1522
+ "grad_norm": 0.295538067817688,
1523
+ "learning_rate": 0.0004085715828952621,
1524
+ "loss": 3.4930191040039062,
1525
+ "step": 19200
1526
+ },
1527
+ {
1528
+ "epoch": 0.386,
1529
+ "grad_norm": 0.18652473390102386,
1530
+ "learning_rate": 0.00040680696748226293,
1531
+ "loss": 3.4893539428710936,
1532
+ "step": 19300
1533
+ },
1534
+ {
1535
+ "epoch": 0.388,
1536
+ "grad_norm": 0.19307976961135864,
1537
+ "learning_rate": 0.00040503811009951325,
1538
+ "loss": 3.484967041015625,
1539
+ "step": 19400
1540
+ },
1541
+ {
1542
+ "epoch": 0.39,
1543
+ "grad_norm": 0.20981250703334808,
1544
+ "learning_rate": 0.0004032650809993542,
1545
+ "loss": 3.4780728149414064,
1546
+ "step": 19500
1547
+ },
1548
+ {
1549
+ "epoch": 0.392,
1550
+ "grad_norm": 0.31385481357574463,
1551
+ "learning_rate": 0.00040148795059981155,
1552
+ "loss": 3.496181640625,
1553
+ "step": 19600
1554
+ },
1555
+ {
1556
+ "epoch": 0.394,
1557
+ "grad_norm": 0.1942823827266693,
1558
+ "learning_rate": 0.0003997067894817995,
1559
+ "loss": 3.487859802246094,
1560
+ "step": 19700
1561
+ },
1562
+ {
1563
+ "epoch": 0.396,
1564
+ "grad_norm": 0.1936560422182083,
1565
+ "learning_rate": 0.0003979216683863172,
1566
+ "loss": 3.479795227050781,
1567
+ "step": 19800
1568
+ },
1569
+ {
1570
+ "epoch": 0.398,
1571
+ "grad_norm": 0.2430066615343094,
1572
+ "learning_rate": 0.00039613265821163883,
1573
+ "loss": 3.4654913330078125,
1574
+ "step": 19900
1575
+ },
1576
+ {
1577
+ "epoch": 0.4,
1578
+ "grad_norm": 0.20538495481014252,
1579
+ "learning_rate": 0.0003943398300104985,
1580
+ "loss": 3.4708984375,
1581
+ "step": 20000
1582
+ },
1583
+ {
1584
+ "epoch": 0.4,
1585
+ "eval_accuracy": 0.36907962990408844,
1586
+ "eval_loss": 3.415916919708252,
1587
+ "eval_runtime": 5.8841,
1588
+ "eval_samples_per_second": 329.872,
1589
+ "eval_steps_per_second": 20.734,
1590
+ "step": 20000
1591
  }
1592
  ],
1593
  "logging_steps": 100,
 
1607
  "attributes": {}
1608
  }
1609
  },
1610
+ "total_flos": 6.271008768e+17,
1611
  "train_batch_size": 120,
1612
  "trial_name": null,
1613
  "trial_params": null