aixk commited on
Commit
84721bd
·
verified ·
1 Parent(s): ae4459d

Upload folder using huggingface_hub

Browse files
checkpoint-800/hybrid_tokenizer_mapping.json CHANGED
The diff for this file is too large to render. See raw diff
 
checkpoint-800/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b924d546b2d6e9f0c3e794b5d6af38a078d3488f324c6d95b1f62fdcf5c49ec5
3
  size 501759848
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:14acfe5d0a60ad01859235191f46b05c380259a3ae31692d5b97fb19370f7ab9
3
  size 501759848
checkpoint-800/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4017f7937b70cba45805cc7651f9675c2d03e2353f9b33d79f0fd87a4e9459db
3
  size 254410983
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6f820a60232165ebc3b04cc7f3ff1672d7999e5857c1c9ce371d4dbb4d1b81e4
3
  size 254410983
checkpoint-800/trainer_state.json CHANGED
@@ -501,72 +501,72 @@
501
  },
502
  {
503
  "epoch": 0.4369062728356602,
504
- "grad_norm": 41.11359786987305,
505
  "learning_rate": 7.02e-05,
506
  "loss": 5.6157,
507
  "step": 710
508
  },
509
  {
510
  "epoch": 0.44305988231221877,
511
- "grad_norm": 46.67518997192383,
512
  "learning_rate": 7.12e-05,
513
  "loss": 5.5794,
514
  "step": 720
515
  },
516
  {
517
  "epoch": 0.44921349178877734,
518
- "grad_norm": 33.07991409301758,
519
  "learning_rate": 7.219999999999999e-05,
520
- "loss": 5.573,
521
  "step": 730
522
  },
523
  {
524
  "epoch": 0.45536710126533597,
525
- "grad_norm": 42.006980895996094,
526
  "learning_rate": 7.319999999999999e-05,
527
  "loss": 5.5528,
528
  "step": 740
529
  },
530
  {
531
  "epoch": 0.46152071074189455,
532
- "grad_norm": 48.60676956176758,
533
  "learning_rate": 7.419999999999999e-05,
534
- "loss": 5.539,
535
  "step": 750
536
  },
537
  {
538
  "epoch": 0.4676743202184531,
539
- "grad_norm": 37.12602996826172,
540
  "learning_rate": 7.519999999999998e-05,
541
  "loss": 5.5043,
542
  "step": 760
543
  },
544
  {
545
  "epoch": 0.47382792969501175,
546
- "grad_norm": 34.83266830444336,
547
  "learning_rate": 7.62e-05,
548
- "loss": 5.4856,
549
  "step": 770
550
  },
551
  {
552
  "epoch": 0.4799815391715703,
553
- "grad_norm": 33.03480911254883,
554
  "learning_rate": 7.72e-05,
555
  "loss": 5.4737,
556
  "step": 780
557
  },
558
  {
559
  "epoch": 0.4861351486481289,
560
- "grad_norm": 36.85569381713867,
561
  "learning_rate": 7.819999999999999e-05,
562
- "loss": 5.469,
563
  "step": 790
564
  },
565
  {
566
  "epoch": 0.49228875812468753,
567
- "grad_norm": 40.46368408203125,
568
  "learning_rate": 7.92e-05,
569
- "loss": 5.4437,
570
  "step": 800
571
  }
572
  ],
 
501
  },
502
  {
503
  "epoch": 0.4369062728356602,
504
+ "grad_norm": 41.100425720214844,
505
  "learning_rate": 7.02e-05,
506
  "loss": 5.6157,
507
  "step": 710
508
  },
509
  {
510
  "epoch": 0.44305988231221877,
511
+ "grad_norm": 46.8226432800293,
512
  "learning_rate": 7.12e-05,
513
  "loss": 5.5794,
514
  "step": 720
515
  },
516
  {
517
  "epoch": 0.44921349178877734,
518
+ "grad_norm": 33.2792854309082,
519
  "learning_rate": 7.219999999999999e-05,
520
+ "loss": 5.5731,
521
  "step": 730
522
  },
523
  {
524
  "epoch": 0.45536710126533597,
525
+ "grad_norm": 44.90331268310547,
526
  "learning_rate": 7.319999999999999e-05,
527
  "loss": 5.5528,
528
  "step": 740
529
  },
530
  {
531
  "epoch": 0.46152071074189455,
532
+ "grad_norm": 48.1407356262207,
533
  "learning_rate": 7.419999999999999e-05,
534
+ "loss": 5.5387,
535
  "step": 750
536
  },
537
  {
538
  "epoch": 0.4676743202184531,
539
+ "grad_norm": 37.969139099121094,
540
  "learning_rate": 7.519999999999998e-05,
541
  "loss": 5.5043,
542
  "step": 760
543
  },
544
  {
545
  "epoch": 0.47382792969501175,
546
+ "grad_norm": 35.248779296875,
547
  "learning_rate": 7.62e-05,
548
+ "loss": 5.4854,
549
  "step": 770
550
  },
551
  {
552
  "epoch": 0.4799815391715703,
553
+ "grad_norm": 32.92762756347656,
554
  "learning_rate": 7.72e-05,
555
  "loss": 5.4737,
556
  "step": 780
557
  },
558
  {
559
  "epoch": 0.4861351486481289,
560
+ "grad_norm": 44.820735931396484,
561
  "learning_rate": 7.819999999999999e-05,
562
+ "loss": 5.4688,
563
  "step": 790
564
  },
565
  {
566
  "epoch": 0.49228875812468753,
567
+ "grad_norm": 46.39222717285156,
568
  "learning_rate": 7.92e-05,
569
+ "loss": 5.4436,
570
  "step": 800
571
  }
572
  ],