CodeIsAbstract commited on
Commit
0fa42be
·
verified ·
1 Parent(s): 586ad79

Training in progress, step 1500, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:87399fda4401e43fddc5a839ada69ffdd168ab918f09d5782bc08b2a08c57590
3
  size 847599616
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bc8d4620e45e3023852d6db363117d2149bf91809954eab034e0f3df6b2e9a53
3
  size 847599616
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:adf30444ff48bb3f968aa531132e86af74895e6dca092ef0b50a37cd281149ea
3
  size 1386414411
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5bbab7f145dce1b83f41efbeae326f52ae177bc04e5b4bcb5acd73f2e563e608
3
  size 1386414411
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3fc325c0f6e133b30ed7ac345224f75619e1ffa24494dab2a3e09b19384d35df
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2b0d1a4257bdb1e850f5a1f4f1d426b1aff1480f1f1be49a43a9b59e27d1626c
3
  size 14917
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:17d95b86c212e1cc1919367792abbcc28abe2eeeaccf2008cdfbf60e31bc1314
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9156180c20e717a1c296208930ef03eb900f1c8749f9b5001b6e2a8c9e24d691
3
  size 14917
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dfdb9865d69d1265ac95b637888c2cf8ac3b8d1ad5a1b9e4f1da654c1dc9106a
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1eb3b2f44ae75601267d4a6c3582add2f5156f5a074e6f3b9a7a8115e10aac03
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.5714285714285714,
6
  "eval_steps": 150,
7
- "global_step": 1200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -920,6 +920,234 @@
920
  "eval_samples_per_second": 19.704,
921
  "eval_steps_per_second": 2.483,
922
  "step": 1200
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
923
  }
924
  ],
925
  "logging_steps": 10,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.7142857142857143,
6
  "eval_steps": 150,
7
+ "global_step": 1500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
920
  "eval_samples_per_second": 19.704,
921
  "eval_steps_per_second": 2.483,
922
  "step": 1200
923
+ },
924
+ {
925
+ "epoch": 0.5761904761904761,
926
+ "grad_norm": 16.894689559936523,
927
+ "learning_rate": 1.6661519745736042e-06,
928
+ "loss": 16.504466247558593,
929
+ "step": 1210
930
+ },
931
+ {
932
+ "epoch": 0.580952380952381,
933
+ "grad_norm": 8.992559432983398,
934
+ "learning_rate": 1.6351418650523748e-06,
935
+ "loss": 17.111102294921874,
936
+ "step": 1220
937
+ },
938
+ {
939
+ "epoch": 0.5857142857142857,
940
+ "grad_norm": 10.779234886169434,
941
+ "learning_rate": 1.6042222306171245e-06,
942
+ "loss": 16.793231201171874,
943
+ "step": 1230
944
+ },
945
+ {
946
+ "epoch": 0.5904761904761905,
947
+ "grad_norm": 13.19653606414795,
948
+ "learning_rate": 1.5734007385125066e-06,
949
+ "loss": 16.616178894042967,
950
+ "step": 1240
951
+ },
952
+ {
953
+ "epoch": 0.5952380952380952,
954
+ "grad_norm": 10.069211959838867,
955
+ "learning_rate": 1.5426850316464914e-06,
956
+ "loss": 16.469076538085936,
957
+ "step": 1250
958
+ },
959
+ {
960
+ "epoch": 0.6,
961
+ "grad_norm": 11.769493103027344,
962
+ "learning_rate": 1.5120827266951346e-06,
963
+ "loss": 16.354193115234374,
964
+ "step": 1260
965
+ },
966
+ {
967
+ "epoch": 0.6047619047619047,
968
+ "grad_norm": 18.091075897216797,
969
+ "learning_rate": 1.4816014122138387e-06,
970
+ "loss": 16.668800354003906,
971
+ "step": 1270
972
+ },
973
+ {
974
+ "epoch": 0.6095238095238096,
975
+ "grad_norm": 40.2747802734375,
976
+ "learning_rate": 1.4512486467555996e-06,
977
+ "loss": 16.58080596923828,
978
+ "step": 1280
979
+ },
980
+ {
981
+ "epoch": 0.6142857142857143,
982
+ "grad_norm": 38.875709533691406,
983
+ "learning_rate": 1.4210319569966813e-06,
984
+ "loss": 16.504058837890625,
985
+ "step": 1290
986
+ },
987
+ {
988
+ "epoch": 0.6190476190476191,
989
+ "grad_norm": 20.455795288085938,
990
+ "learning_rate": 1.3909588358702065e-06,
991
+ "loss": 16.619796752929688,
992
+ "step": 1300
993
+ },
994
+ {
995
+ "epoch": 0.6238095238095238,
996
+ "grad_norm": 26.14783477783203,
997
+ "learning_rate": 1.3610367407081037e-06,
998
+ "loss": 16.69707336425781,
999
+ "step": 1310
1000
+ },
1001
+ {
1002
+ "epoch": 0.6285714285714286,
1003
+ "grad_norm": 12.17022705078125,
1004
+ "learning_rate": 1.3312730913918927e-06,
1005
+ "loss": 16.7201904296875,
1006
+ "step": 1320
1007
+ },
1008
+ {
1009
+ "epoch": 0.6333333333333333,
1010
+ "grad_norm": 13.315362930297852,
1011
+ "learning_rate": 1.3016752685127483e-06,
1012
+ "loss": 16.331112670898438,
1013
+ "step": 1330
1014
+ },
1015
+ {
1016
+ "epoch": 0.638095238095238,
1017
+ "grad_norm": 6.595879077911377,
1018
+ "learning_rate": 1.2722506115413118e-06,
1019
+ "loss": 16.394076538085937,
1020
+ "step": 1340
1021
+ },
1022
+ {
1023
+ "epoch": 0.6428571428571429,
1024
+ "grad_norm": 7.89982271194458,
1025
+ "learning_rate": 1.243006417007699e-06,
1026
+ "loss": 16.70445098876953,
1027
+ "step": 1350
1028
+ },
1029
+ {
1030
+ "epoch": 0.6428571428571429,
1031
+ "eval_accuracy": 0.06341732283464567,
1032
+ "eval_loss": 16.499183654785156,
1033
+ "eval_runtime": 25.1886,
1034
+ "eval_samples_per_second": 19.85,
1035
+ "eval_steps_per_second": 2.501,
1036
+ "step": 1350
1037
+ },
1038
+ {
1039
+ "epoch": 0.6476190476190476,
1040
+ "grad_norm": 9.356073379516602,
1041
+ "learning_rate": 1.2139499366921528e-06,
1042
+ "loss": 16.716755676269532,
1043
+ "step": 1360
1044
+ },
1045
+ {
1046
+ "epoch": 0.6523809523809524,
1047
+ "grad_norm": 13.010209083557129,
1048
+ "learning_rate": 1.185088375826799e-06,
1049
+ "loss": 17.275030517578124,
1050
+ "step": 1370
1051
+ },
1052
+ {
1053
+ "epoch": 0.6571428571428571,
1054
+ "grad_norm": 7.622005462646484,
1055
+ "learning_rate": 1.156428891308936e-06,
1056
+ "loss": 16.587196350097656,
1057
+ "step": 1380
1058
+ },
1059
+ {
1060
+ "epoch": 0.6619047619047619,
1061
+ "grad_norm": 11.086586952209473,
1062
+ "learning_rate": 1.1279785899263192e-06,
1063
+ "loss": 16.67912902832031,
1064
+ "step": 1390
1065
+ },
1066
+ {
1067
+ "epoch": 0.6666666666666666,
1068
+ "grad_norm": 22.962263107299805,
1069
+ "learning_rate": 1.099744526594867e-06,
1070
+ "loss": 16.530209350585938,
1071
+ "step": 1400
1072
+ },
1073
+ {
1074
+ "epoch": 0.6714285714285714,
1075
+ "grad_norm": 7.309661388397217,
1076
+ "learning_rate": 1.0717337026092269e-06,
1077
+ "loss": 16.585041809082032,
1078
+ "step": 1410
1079
+ },
1080
+ {
1081
+ "epoch": 0.6761904761904762,
1082
+ "grad_norm": 7.678780555725098,
1083
+ "learning_rate": 1.0439530639066418e-06,
1084
+ "loss": 16.243795776367186,
1085
+ "step": 1420
1086
+ },
1087
+ {
1088
+ "epoch": 0.680952380952381,
1089
+ "grad_norm": 38.843414306640625,
1090
+ "learning_rate": 1.0164094993445473e-06,
1091
+ "loss": 16.31683349609375,
1092
+ "step": 1430
1093
+ },
1094
+ {
1095
+ "epoch": 0.6857142857142857,
1096
+ "grad_norm": 8.149712562561035,
1097
+ "learning_rate": 9.891098389923132e-07,
1098
+ "loss": 16.380844116210938,
1099
+ "step": 1440
1100
+ },
1101
+ {
1102
+ "epoch": 0.6904761904761905,
1103
+ "grad_norm": 23.914464950561523,
1104
+ "learning_rate": 9.620608524375702e-07,
1105
+ "loss": 16.283917236328126,
1106
+ "step": 1450
1107
+ },
1108
+ {
1109
+ "epoch": 0.6952380952380952,
1110
+ "grad_norm": 17.59719467163086,
1111
+ "learning_rate": 9.352692471075336e-07,
1112
+ "loss": 16.381610107421874,
1113
+ "step": 1460
1114
+ },
1115
+ {
1116
+ "epoch": 0.7,
1117
+ "grad_norm": 56.93424606323242,
1118
+ "learning_rate": 9.087416666057417e-07,
1119
+ "loss": 17.249729919433594,
1120
+ "step": 1470
1121
+ },
1122
+ {
1123
+ "epoch": 0.7047619047619048,
1124
+ "grad_norm": 9.12337589263916,
1125
+ "learning_rate": 8.824846890646149e-07,
1126
+ "loss": 16.201866149902344,
1127
+ "step": 1480
1128
+ },
1129
+ {
1130
+ "epoch": 0.7095238095238096,
1131
+ "grad_norm": 10.127345085144043,
1132
+ "learning_rate": 8.565048255142577e-07,
1133
+ "loss": 16.494973754882814,
1134
+ "step": 1490
1135
+ },
1136
+ {
1137
+ "epoch": 0.7142857142857143,
1138
+ "grad_norm": 23.02049446105957,
1139
+ "learning_rate": 8.308085182678972e-07,
1140
+ "loss": 16.431625366210938,
1141
+ "step": 1500
1142
+ },
1143
+ {
1144
+ "epoch": 0.7142857142857143,
1145
+ "eval_accuracy": 0.06583464566929134,
1146
+ "eval_loss": 16.394376754760742,
1147
+ "eval_runtime": 25.1159,
1148
+ "eval_samples_per_second": 19.908,
1149
+ "eval_steps_per_second": 2.508,
1150
+ "step": 1500
1151
  }
1152
  ],
1153
  "logging_steps": 10,