CodeIsAbstract commited on
Commit
c4bc9de
·
verified ·
1 Parent(s): 44873d7

Training in progress, step 36000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fc3c5b4f1c32d282710a153ae5bf5ca6bd7cbd3f4f67f12a3907a6cd19d2d279
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:113fb894b062b29cf117cc658db504261451a8ff8c1a0cb06fb7e28935cf14f7
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:605c9b21acde75c1394f97c25a318d3192f035d829b96871f4ced0b57206ec1c
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a46699af39339744e0585d58fd78b70cf3632fc84e8d7d9301d7e78f3c4dfc81
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6177091d3805fe7255fbeea64136d688b15629fc946401f1853fbd6e6a2f18ad
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cadea5b42974dfb2cd316f1afeedc6d7ac097ea0c6cf0e13d2fffcae3de277b9
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:30eae1c116a1408bbe4aec95a1368b13c0bf69d349ca373bf4a8fd3a1e9fc110
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f1f70882941e6b31d82263e4fc807f24daa03205e781becc1cf39b5843e1bb88
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.2909090909090909,
6
  "eval_steps": 1000,
7
- "global_step": 32000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2504,6 +2504,318 @@
2504
  "eval_samples_per_second": 84.308,
2505
  "eval_steps_per_second": 21.077,
2506
  "step": 32000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2507
  }
2508
  ],
2509
  "logging_steps": 100,
@@ -2523,7 +2835,7 @@
2523
  "attributes": {}
2524
  }
2525
  },
2526
- "total_flos": 7.97266591875072e+17,
2527
  "train_batch_size": 22,
2528
  "trial_name": null,
2529
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.03636363636363636,
6
  "eval_steps": 1000,
7
+ "global_step": 36000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2504
  "eval_samples_per_second": 84.308,
2505
  "eval_steps_per_second": 21.077,
2506
  "step": 32000
2507
+ },
2508
+ {
2509
+ "epoch": 0.0009090909090909091,
2510
+ "grad_norm": 0.22038772702217102,
2511
+ "learning_rate": 0.0008054073172385756,
2512
+ "loss": 2.794765625,
2513
+ "step": 32100
2514
+ },
2515
+ {
2516
+ "epoch": 0.0018181818181818182,
2517
+ "grad_norm": 0.15077297389507294,
2518
+ "learning_rate": 0.0008042738758196204,
2519
+ "loss": 2.7949072265625,
2520
+ "step": 32200
2521
+ },
2522
+ {
2523
+ "epoch": 0.0027272727272727275,
2524
+ "grad_norm": 0.1514005959033966,
2525
+ "learning_rate": 0.0008031379457496854,
2526
+ "loss": 2.8156253051757814,
2527
+ "step": 32300
2528
+ },
2529
+ {
2530
+ "epoch": 0.0036363636363636364,
2531
+ "grad_norm": 0.13997957110404968,
2532
+ "learning_rate": 0.0008019995363195234,
2533
+ "loss": 2.8109649658203124,
2534
+ "step": 32400
2535
+ },
2536
+ {
2537
+ "epoch": 0.004545454545454545,
2538
+ "grad_norm": 0.14847084879875183,
2539
+ "learning_rate": 0.0008008586568401662,
2540
+ "loss": 2.7742608642578124,
2541
+ "step": 32500
2542
+ },
2543
+ {
2544
+ "epoch": 0.005454545454545455,
2545
+ "grad_norm": 0.13216467201709747,
2546
+ "learning_rate": 0.0007997153166428485,
2547
+ "loss": 2.7673321533203126,
2548
+ "step": 32600
2549
+ },
2550
+ {
2551
+ "epoch": 0.006363636363636364,
2552
+ "grad_norm": 0.16374439001083374,
2553
+ "learning_rate": 0.0007985695250789306,
2554
+ "loss": 2.78884521484375,
2555
+ "step": 32700
2556
+ },
2557
+ {
2558
+ "epoch": 0.007272727272727273,
2559
+ "grad_norm": 0.1394331306219101,
2560
+ "learning_rate": 0.0007974212915198225,
2561
+ "loss": 2.779542236328125,
2562
+ "step": 32800
2563
+ },
2564
+ {
2565
+ "epoch": 0.008181818181818182,
2566
+ "grad_norm": 0.1565932035446167,
2567
+ "learning_rate": 0.0007962706253569077,
2568
+ "loss": 2.815880126953125,
2569
+ "step": 32900
2570
+ },
2571
+ {
2572
+ "epoch": 0.00909090909090909,
2573
+ "grad_norm": 0.17284399271011353,
2574
+ "learning_rate": 0.0007951175360014656,
2575
+ "loss": 2.769788818359375,
2576
+ "step": 33000
2577
+ },
2578
+ {
2579
+ "epoch": 0.00909090909090909,
2580
+ "eval_loss": 3.1402082443237305,
2581
+ "eval_runtime": 7.4554,
2582
+ "eval_samples_per_second": 77.259,
2583
+ "eval_steps_per_second": 19.315,
2584
+ "step": 33000
2585
+ },
2586
+ {
2587
+ "epoch": 0.01,
2588
+ "grad_norm": 0.15111324191093445,
2589
+ "learning_rate": 0.0007939620328845948,
2590
+ "loss": 2.7730609130859376,
2591
+ "step": 33100
2592
+ },
2593
+ {
2594
+ "epoch": 0.01090909090909091,
2595
+ "grad_norm": 0.160688117146492,
2596
+ "learning_rate": 0.0007928041254571361,
2597
+ "loss": 2.787190246582031,
2598
+ "step": 33200
2599
+ },
2600
+ {
2601
+ "epoch": 0.011818181818181818,
2602
+ "grad_norm": 0.17507240176200867,
2603
+ "learning_rate": 0.0007916438231895952,
2604
+ "loss": 2.766163330078125,
2605
+ "step": 33300
2606
+ },
2607
+ {
2608
+ "epoch": 0.012727272727272728,
2609
+ "grad_norm": 0.15393611788749695,
2610
+ "learning_rate": 0.0007904811355720652,
2611
+ "loss": 2.7740924072265627,
2612
+ "step": 33400
2613
+ },
2614
+ {
2615
+ "epoch": 0.013636363636363636,
2616
+ "grad_norm": 0.14618468284606934,
2617
+ "learning_rate": 0.0007893160721141486,
2618
+ "loss": 2.7595263671875,
2619
+ "step": 33500
2620
+ },
2621
+ {
2622
+ "epoch": 0.014545454545454545,
2623
+ "grad_norm": 0.1701638251543045,
2624
+ "learning_rate": 0.0007881486423448802,
2625
+ "loss": 2.7771051025390623,
2626
+ "step": 33600
2627
+ },
2628
+ {
2629
+ "epoch": 0.015454545454545455,
2630
+ "grad_norm": 0.15899600088596344,
2631
+ "learning_rate": 0.0007869788558126487,
2632
+ "loss": 2.754100646972656,
2633
+ "step": 33700
2634
+ },
2635
+ {
2636
+ "epoch": 0.016363636363636365,
2637
+ "grad_norm": 0.14573192596435547,
2638
+ "learning_rate": 0.0007858067220851188,
2639
+ "loss": 2.7869033813476562,
2640
+ "step": 33800
2641
+ },
2642
+ {
2643
+ "epoch": 0.017272727272727273,
2644
+ "grad_norm": 0.16796821355819702,
2645
+ "learning_rate": 0.0007846322507491528,
2646
+ "loss": 2.7748208618164063,
2647
+ "step": 33900
2648
+ },
2649
+ {
2650
+ "epoch": 0.01818181818181818,
2651
+ "grad_norm": 0.1425439566373825,
2652
+ "learning_rate": 0.000783455451410732,
2653
+ "loss": 2.773130187988281,
2654
+ "step": 34000
2655
+ },
2656
+ {
2657
+ "epoch": 0.01818181818181818,
2658
+ "eval_loss": 3.13128399848938,
2659
+ "eval_runtime": 7.4414,
2660
+ "eval_samples_per_second": 77.405,
2661
+ "eval_steps_per_second": 19.351,
2662
+ "step": 34000
2663
+ },
2664
+ {
2665
+ "epoch": 0.019090909090909092,
2666
+ "grad_norm": 0.15474802255630493,
2667
+ "learning_rate": 0.000782276333694879,
2668
+ "loss": 2.7699960327148436,
2669
+ "step": 34100
2670
+ },
2671
+ {
2672
+ "epoch": 0.02,
2673
+ "grad_norm": 0.1559676229953766,
2674
+ "learning_rate": 0.0007810949072455778,
2675
+ "loss": 2.7731719970703126,
2676
+ "step": 34200
2677
+ },
2678
+ {
2679
+ "epoch": 0.02090909090909091,
2680
+ "grad_norm": 0.1522807478904724,
2681
+ "learning_rate": 0.0007799111817256959,
2682
+ "loss": 2.7761361694335935,
2683
+ "step": 34300
2684
+ },
2685
+ {
2686
+ "epoch": 0.02181818181818182,
2687
+ "grad_norm": 0.15091344714164734,
2688
+ "learning_rate": 0.0007787251668169043,
2689
+ "loss": 2.7527685546875,
2690
+ "step": 34400
2691
+ },
2692
+ {
2693
+ "epoch": 0.022727272727272728,
2694
+ "grad_norm": 0.15074443817138672,
2695
+ "learning_rate": 0.0007775368722195997,
2696
+ "loss": 2.7675704956054688,
2697
+ "step": 34500
2698
+ },
2699
+ {
2700
+ "epoch": 0.023636363636363636,
2701
+ "grad_norm": 0.1461259126663208,
2702
+ "learning_rate": 0.0007763463076528237,
2703
+ "loss": 2.76167236328125,
2704
+ "step": 34600
2705
+ },
2706
+ {
2707
+ "epoch": 0.024545454545454544,
2708
+ "grad_norm": 0.15030233561992645,
2709
+ "learning_rate": 0.0007751534828541842,
2710
+ "loss": 2.7275946044921877,
2711
+ "step": 34700
2712
+ },
2713
+ {
2714
+ "epoch": 0.025454545454545455,
2715
+ "grad_norm": 0.14424681663513184,
2716
+ "learning_rate": 0.0007739584075797753,
2717
+ "loss": 2.736708679199219,
2718
+ "step": 34800
2719
+ },
2720
+ {
2721
+ "epoch": 0.026363636363636363,
2722
+ "grad_norm": 0.15576180815696716,
2723
+ "learning_rate": 0.0007727610916040981,
2724
+ "loss": 2.7289297485351565,
2725
+ "step": 34900
2726
+ },
2727
+ {
2728
+ "epoch": 0.02727272727272727,
2729
+ "grad_norm": 0.1324995905160904,
2730
+ "learning_rate": 0.0007715615447199798,
2731
+ "loss": 2.7693435668945314,
2732
+ "step": 35000
2733
+ },
2734
+ {
2735
+ "epoch": 0.02727272727272727,
2736
+ "eval_loss": 3.1262025833129883,
2737
+ "eval_runtime": 7.4365,
2738
+ "eval_samples_per_second": 77.456,
2739
+ "eval_steps_per_second": 19.364,
2740
+ "step": 35000
2741
+ },
2742
+ {
2743
+ "epoch": 0.028181818181818183,
2744
+ "grad_norm": 0.1613311767578125,
2745
+ "learning_rate": 0.0007703597767384945,
2746
+ "loss": 2.7539468383789063,
2747
+ "step": 35100
2748
+ },
2749
+ {
2750
+ "epoch": 0.02909090909090909,
2751
+ "grad_norm": 0.16793298721313477,
2752
+ "learning_rate": 0.0007691557974888825,
2753
+ "loss": 2.7785479736328127,
2754
+ "step": 35200
2755
+ },
2756
+ {
2757
+ "epoch": 0.03,
2758
+ "grad_norm": 0.14937831461429596,
2759
+ "learning_rate": 0.0007679496168184705,
2760
+ "loss": 2.7587347412109375,
2761
+ "step": 35300
2762
+ },
2763
+ {
2764
+ "epoch": 0.03090909090909091,
2765
+ "grad_norm": 0.1619419902563095,
2766
+ "learning_rate": 0.0007667412445925898,
2767
+ "loss": 2.7732388305664064,
2768
+ "step": 35400
2769
+ },
2770
+ {
2771
+ "epoch": 0.031818181818181815,
2772
+ "grad_norm": 0.15758360922336578,
2773
+ "learning_rate": 0.000765530690694497,
2774
+ "loss": 2.74038818359375,
2775
+ "step": 35500
2776
+ },
2777
+ {
2778
+ "epoch": 0.03272727272727273,
2779
+ "grad_norm": 0.1523827463388443,
2780
+ "learning_rate": 0.000764317965025292,
2781
+ "loss": 2.7354312133789063,
2782
+ "step": 35600
2783
+ },
2784
+ {
2785
+ "epoch": 0.03363636363636364,
2786
+ "grad_norm": 0.1685842126607895,
2787
+ "learning_rate": 0.0007631030775038383,
2788
+ "loss": 2.7525250244140627,
2789
+ "step": 35700
2790
+ },
2791
+ {
2792
+ "epoch": 0.034545454545454546,
2793
+ "grad_norm": 0.1396898627281189,
2794
+ "learning_rate": 0.0007618860380666808,
2795
+ "loss": 2.737476501464844,
2796
+ "step": 35800
2797
+ },
2798
+ {
2799
+ "epoch": 0.035454545454545454,
2800
+ "grad_norm": 0.15296360850334167,
2801
+ "learning_rate": 0.0007606668566679644,
2802
+ "loss": 2.747691955566406,
2803
+ "step": 35900
2804
+ },
2805
+ {
2806
+ "epoch": 0.03636363636363636,
2807
+ "grad_norm": 0.1561899036169052,
2808
+ "learning_rate": 0.0007594455432793539,
2809
+ "loss": 2.7320767211914063,
2810
+ "step": 36000
2811
+ },
2812
+ {
2813
+ "epoch": 0.03636363636363636,
2814
+ "eval_loss": 3.1245434284210205,
2815
+ "eval_runtime": 7.4489,
2816
+ "eval_samples_per_second": 77.327,
2817
+ "eval_steps_per_second": 19.332,
2818
+ "step": 36000
2819
  }
2820
  ],
2821
  "logging_steps": 100,
 
2835
  "attributes": {}
2836
  }
2837
  },
2838
+ "total_flos": 8.96924915859456e+17,
2839
  "train_batch_size": 22,
2840
  "trial_name": null,
2841
  "trial_params": null
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:567e8406e27dd015c52febc5cd0c60d00147e3fcfbac62fe7f312af41586ea47
3
  size 5201
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ef8c80f5d40636555b8c09ec97785c8964e446b8d757c55eb76e5b7454841939
3
  size 5201