kiritan commited on
Commit
4327723
·
verified ·
1 Parent(s): 2e61bf9

Training in progress, step 4000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:98a40d1f879e92b1250c6ce6814b61dc6d1fca0c70ebd417d8ead38a7dce3f4e
3
  size 223144592
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a1695d6ce8bd5b7d2355edbbd0898111b45ff3435a1ab08b8365805bc3f3ad00
3
  size 223144592
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:77c69b1b51235eeb3ea13d92f8c6dcf9f655d0d2d123b4b47cd7aef8a879fb06
3
  size 440235130
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f12dff49781fca74d122378f26745ac52fd4b91b3decbb3b980a33448a5411b5
3
  size 440235130
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4bf29b98658cbaec0d62f4ead6c42ba9b88966fa3b2ce6348ec1b4bf21ab3255
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:da7113ba4a15f803acd596486d291dff44a22010c3be333f82da6ae49ecda7bd
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:767f5d686c361afdbfc4427661911446768c0b14e80b00b46139bfc4c493d9cb
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:21d3eeef439eb9e15103ee606f4a4606b33da69a5d17044fd608528bcb1583fb
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
- "best_metric": 96.64309288071664,
3
- "best_model_checkpoint": "./iteboshi_temp/checkpoint-3000",
4
- "epoch": 4.951279933938894,
5
  "eval_steps": 1000,
6
- "global_step": 3000,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -877,6 +877,296 @@
877
  "eval_steps_per_second": 1.17,
878
  "eval_wer": 96.64309288071664,
879
  "step": 3000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
880
  }
881
  ],
882
  "logging_steps": 25,
@@ -896,7 +1186,7 @@
896
  "attributes": {}
897
  }
898
  },
899
- "total_flos": 5.84849320574976e+18,
900
  "train_batch_size": 12,
901
  "trial_name": null,
902
  "trial_params": null
 
1
  {
2
+ "best_metric": 95.4926921263555,
3
+ "best_model_checkpoint": "./iteboshi_temp/checkpoint-4000",
4
+ "epoch": 6.601156069364162,
5
  "eval_steps": 1000,
6
+ "global_step": 4000,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
877
  "eval_steps_per_second": 1.17,
878
  "eval_wer": 96.64309288071664,
879
  "step": 3000
880
+ },
881
+ {
882
+ "epoch": 4.992568125516103,
883
+ "grad_norm": 7.915883541107178,
884
+ "learning_rate": 1.4686315789473687e-05,
885
+ "loss": 0.7496,
886
+ "step": 3025
887
+ },
888
+ {
889
+ "epoch": 5.033030553261767,
890
+ "grad_norm": 7.240871906280518,
891
+ "learning_rate": 1.4633684210526317e-05,
892
+ "loss": 0.687,
893
+ "step": 3050
894
+ },
895
+ {
896
+ "epoch": 5.074318744838976,
897
+ "grad_norm": 7.480823516845703,
898
+ "learning_rate": 1.4581052631578949e-05,
899
+ "loss": 0.6601,
900
+ "step": 3075
901
+ },
902
+ {
903
+ "epoch": 5.115606936416185,
904
+ "grad_norm": 7.587548732757568,
905
+ "learning_rate": 1.452842105263158e-05,
906
+ "loss": 0.6831,
907
+ "step": 3100
908
+ },
909
+ {
910
+ "epoch": 5.156895127993394,
911
+ "grad_norm": 7.099322319030762,
912
+ "learning_rate": 1.4475789473684212e-05,
913
+ "loss": 0.6658,
914
+ "step": 3125
915
+ },
916
+ {
917
+ "epoch": 5.198183319570603,
918
+ "grad_norm": 7.174043655395508,
919
+ "learning_rate": 1.4423157894736843e-05,
920
+ "loss": 0.6704,
921
+ "step": 3150
922
+ },
923
+ {
924
+ "epoch": 5.239471511147812,
925
+ "grad_norm": 6.899098873138428,
926
+ "learning_rate": 1.4370526315789475e-05,
927
+ "loss": 0.65,
928
+ "step": 3175
929
+ },
930
+ {
931
+ "epoch": 5.280759702725021,
932
+ "grad_norm": 7.519511699676514,
933
+ "learning_rate": 1.4317894736842107e-05,
934
+ "loss": 0.6665,
935
+ "step": 3200
936
+ },
937
+ {
938
+ "epoch": 5.32204789430223,
939
+ "grad_norm": 7.408304214477539,
940
+ "learning_rate": 1.4265263157894738e-05,
941
+ "loss": 0.6398,
942
+ "step": 3225
943
+ },
944
+ {
945
+ "epoch": 5.363336085879438,
946
+ "grad_norm": 7.133185386657715,
947
+ "learning_rate": 1.4212631578947368e-05,
948
+ "loss": 0.6802,
949
+ "step": 3250
950
+ },
951
+ {
952
+ "epoch": 5.404624277456647,
953
+ "grad_norm": 7.851741313934326,
954
+ "learning_rate": 1.416e-05,
955
+ "loss": 0.6327,
956
+ "step": 3275
957
+ },
958
+ {
959
+ "epoch": 5.445912469033856,
960
+ "grad_norm": 7.773444175720215,
961
+ "learning_rate": 1.4107368421052632e-05,
962
+ "loss": 0.6998,
963
+ "step": 3300
964
+ },
965
+ {
966
+ "epoch": 5.487200660611065,
967
+ "grad_norm": 8.28067398071289,
968
+ "learning_rate": 1.4054736842105263e-05,
969
+ "loss": 0.6726,
970
+ "step": 3325
971
+ },
972
+ {
973
+ "epoch": 5.528488852188274,
974
+ "grad_norm": 6.886124134063721,
975
+ "learning_rate": 1.4002105263157897e-05,
976
+ "loss": 0.6418,
977
+ "step": 3350
978
+ },
979
+ {
980
+ "epoch": 5.569777043765483,
981
+ "grad_norm": 6.617015361785889,
982
+ "learning_rate": 1.3949473684210528e-05,
983
+ "loss": 0.6613,
984
+ "step": 3375
985
+ },
986
+ {
987
+ "epoch": 5.611065235342692,
988
+ "grad_norm": 7.447840690612793,
989
+ "learning_rate": 1.389684210526316e-05,
990
+ "loss": 0.6608,
991
+ "step": 3400
992
+ },
993
+ {
994
+ "epoch": 5.652353426919901,
995
+ "grad_norm": 7.151592254638672,
996
+ "learning_rate": 1.3844210526315791e-05,
997
+ "loss": 0.6409,
998
+ "step": 3425
999
+ },
1000
+ {
1001
+ "epoch": 5.69364161849711,
1002
+ "grad_norm": 7.587296962738037,
1003
+ "learning_rate": 1.3791578947368423e-05,
1004
+ "loss": 0.6483,
1005
+ "step": 3450
1006
+ },
1007
+ {
1008
+ "epoch": 5.734929810074319,
1009
+ "grad_norm": 7.848781585693359,
1010
+ "learning_rate": 1.3738947368421055e-05,
1011
+ "loss": 0.6638,
1012
+ "step": 3475
1013
+ },
1014
+ {
1015
+ "epoch": 5.776218001651528,
1016
+ "grad_norm": 7.8602986335754395,
1017
+ "learning_rate": 1.3686315789473685e-05,
1018
+ "loss": 0.6464,
1019
+ "step": 3500
1020
+ },
1021
+ {
1022
+ "epoch": 5.817506193228737,
1023
+ "grad_norm": 7.792623043060303,
1024
+ "learning_rate": 1.3633684210526316e-05,
1025
+ "loss": 0.6641,
1026
+ "step": 3525
1027
+ },
1028
+ {
1029
+ "epoch": 5.858794384805946,
1030
+ "grad_norm": 7.796040058135986,
1031
+ "learning_rate": 1.3581052631578948e-05,
1032
+ "loss": 0.6529,
1033
+ "step": 3550
1034
+ },
1035
+ {
1036
+ "epoch": 5.900082576383154,
1037
+ "grad_norm": 7.527121543884277,
1038
+ "learning_rate": 1.352842105263158e-05,
1039
+ "loss": 0.6458,
1040
+ "step": 3575
1041
+ },
1042
+ {
1043
+ "epoch": 5.941370767960363,
1044
+ "grad_norm": 6.546653747558594,
1045
+ "learning_rate": 1.3475789473684211e-05,
1046
+ "loss": 0.6087,
1047
+ "step": 3600
1048
+ },
1049
+ {
1050
+ "epoch": 5.982658959537572,
1051
+ "grad_norm": 8.012529373168945,
1052
+ "learning_rate": 1.3423157894736843e-05,
1053
+ "loss": 0.6298,
1054
+ "step": 3625
1055
+ },
1056
+ {
1057
+ "epoch": 6.023121387283237,
1058
+ "grad_norm": 7.391585350036621,
1059
+ "learning_rate": 1.3370526315789475e-05,
1060
+ "loss": 0.5543,
1061
+ "step": 3650
1062
+ },
1063
+ {
1064
+ "epoch": 6.064409578860446,
1065
+ "grad_norm": 6.664039134979248,
1066
+ "learning_rate": 1.3317894736842108e-05,
1067
+ "loss": 0.5252,
1068
+ "step": 3675
1069
+ },
1070
+ {
1071
+ "epoch": 6.105697770437655,
1072
+ "grad_norm": 6.451517105102539,
1073
+ "learning_rate": 1.326526315789474e-05,
1074
+ "loss": 0.5712,
1075
+ "step": 3700
1076
+ },
1077
+ {
1078
+ "epoch": 6.146985962014864,
1079
+ "grad_norm": 6.391648292541504,
1080
+ "learning_rate": 1.321263157894737e-05,
1081
+ "loss": 0.5664,
1082
+ "step": 3725
1083
+ },
1084
+ {
1085
+ "epoch": 6.188274153592072,
1086
+ "grad_norm": 6.524181842803955,
1087
+ "learning_rate": 1.3160000000000001e-05,
1088
+ "loss": 0.5514,
1089
+ "step": 3750
1090
+ },
1091
+ {
1092
+ "epoch": 6.229562345169281,
1093
+ "grad_norm": 6.788370609283447,
1094
+ "learning_rate": 1.3107368421052633e-05,
1095
+ "loss": 0.5693,
1096
+ "step": 3775
1097
+ },
1098
+ {
1099
+ "epoch": 6.27085053674649,
1100
+ "grad_norm": 6.802549839019775,
1101
+ "learning_rate": 1.3054736842105264e-05,
1102
+ "loss": 0.5332,
1103
+ "step": 3800
1104
+ },
1105
+ {
1106
+ "epoch": 6.312138728323699,
1107
+ "grad_norm": 6.864429473876953,
1108
+ "learning_rate": 1.3002105263157896e-05,
1109
+ "loss": 0.5582,
1110
+ "step": 3825
1111
+ },
1112
+ {
1113
+ "epoch": 6.353426919900908,
1114
+ "grad_norm": 7.168697357177734,
1115
+ "learning_rate": 1.2949473684210528e-05,
1116
+ "loss": 0.5471,
1117
+ "step": 3850
1118
+ },
1119
+ {
1120
+ "epoch": 6.394715111478117,
1121
+ "grad_norm": 6.500988960266113,
1122
+ "learning_rate": 1.289684210526316e-05,
1123
+ "loss": 0.5472,
1124
+ "step": 3875
1125
+ },
1126
+ {
1127
+ "epoch": 6.436003303055326,
1128
+ "grad_norm": 6.378172874450684,
1129
+ "learning_rate": 1.2844210526315791e-05,
1130
+ "loss": 0.5512,
1131
+ "step": 3900
1132
+ },
1133
+ {
1134
+ "epoch": 6.477291494632535,
1135
+ "grad_norm": 6.865861415863037,
1136
+ "learning_rate": 1.279157894736842e-05,
1137
+ "loss": 0.5739,
1138
+ "step": 3925
1139
+ },
1140
+ {
1141
+ "epoch": 6.518579686209744,
1142
+ "grad_norm": 6.929198741912842,
1143
+ "learning_rate": 1.2738947368421052e-05,
1144
+ "loss": 0.5654,
1145
+ "step": 3950
1146
+ },
1147
+ {
1148
+ "epoch": 6.559867877786953,
1149
+ "grad_norm": 6.951797008514404,
1150
+ "learning_rate": 1.2686315789473684e-05,
1151
+ "loss": 0.5422,
1152
+ "step": 3975
1153
+ },
1154
+ {
1155
+ "epoch": 6.601156069364162,
1156
+ "grad_norm": 7.957674980163574,
1157
+ "learning_rate": 1.2633684210526316e-05,
1158
+ "loss": 0.5464,
1159
+ "step": 4000
1160
+ },
1161
+ {
1162
+ "epoch": 6.601156069364162,
1163
+ "eval_cer": 48.25066202010707,
1164
+ "eval_loss": 1.0574802160263062,
1165
+ "eval_runtime": 742.515,
1166
+ "eval_samples_per_second": 14.25,
1167
+ "eval_steps_per_second": 1.188,
1168
+ "eval_wer": 95.4926921263555,
1169
+ "step": 4000
1170
  }
1171
  ],
1172
  "logging_steps": 25,
 
1186
  "attributes": {}
1187
  }
1188
  },
1189
+ "total_flos": 7.79712372670464e+18,
1190
  "train_batch_size": 12,
1191
  "trial_name": null,
1192
  "trial_params": null