CodeIsAbstract commited on
Commit
8d012b7
·
verified ·
1 Parent(s): 8755d2e

Training in progress, step 100, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d9b20be4465fc9f2fdc45a6b03c5b94b7d02f5beb0c6ee41705041ee2fa42bb1
3
  size 847599616
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:66c490ccd97ff3a1c5be26a7a6caff2ab3d0f0f23f0ba25ff8645d1cc81be9dd
3
  size 847599616
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a89a4dd12c905f92cb01197f96241d3b525490721258224cc4c8522e1e53dc47
3
  size 350603
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5ebb2eae727fa1d60dd1ae5c5487d2139dfee54a62ac5572c925eef09f931879
3
  size 350603
last-checkpoint/trainer_state.json CHANGED
@@ -11,160 +11,160 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.025,
14
- "grad_norm": 0.10900866985321045,
15
  "learning_rate": 1.6e-06,
16
- "loss": 11.06564712524414,
17
  "step": 5
18
  },
19
  {
20
  "epoch": 0.05,
21
- "grad_norm": 0.10152705758810043,
22
  "learning_rate": 3.6e-06,
23
- "loss": 11.040735626220703,
24
  "step": 10
25
  },
26
  {
27
  "epoch": 0.075,
28
- "grad_norm": 0.09875829517841339,
29
  "learning_rate": 3.995627254437549e-06,
30
- "loss": 11.026320648193359,
31
  "step": 15
32
  },
33
  {
34
  "epoch": 0.1,
35
- "grad_norm": 0.09957622736692429,
36
  "learning_rate": 3.9778957412029366e-06,
37
- "loss": 11.012971496582031,
38
  "step": 20
39
  },
40
  {
41
  "epoch": 0.125,
42
- "grad_norm": 0.1074257642030716,
43
  "learning_rate": 3.9466531944960116e-06,
44
- "loss": 10.992045593261718,
45
  "step": 25
46
  },
47
  {
48
  "epoch": 0.15,
49
- "grad_norm": 0.110373854637146,
50
  "learning_rate": 3.902113032590307e-06,
51
- "loss": 10.97699432373047,
52
  "step": 30
53
  },
54
  {
55
  "epoch": 0.175,
56
- "grad_norm": 0.11517436057329178,
57
  "learning_rate": 3.844579509954769e-06,
58
- "loss": 10.947827911376953,
59
  "step": 35
60
  },
61
  {
62
  "epoch": 0.2,
63
- "grad_norm": 0.1078442707657814,
64
  "learning_rate": 3.7744456388872234e-06,
65
- "loss": 10.949494171142579,
66
  "step": 40
67
  },
68
  {
69
  "epoch": 0.225,
70
- "grad_norm": 0.1252269148826599,
71
  "learning_rate": 3.6921905048418706e-06,
72
- "loss": 10.916586303710938,
73
  "step": 45
74
  },
75
  {
76
  "epoch": 0.25,
77
- "grad_norm": 0.12218247354030609,
78
  "learning_rate": 3.598375993789849e-06,
79
- "loss": 10.902359008789062,
80
  "step": 50
81
  },
82
  {
83
  "epoch": 0.25,
84
- "eval_accuracy": 0.001978021978021978,
85
- "eval_loss": 10.892147064208984,
86
- "eval_runtime": 11.0414,
87
- "eval_samples_per_second": 9.057,
88
- "eval_steps_per_second": 1.54,
89
  "step": 50
90
  },
91
  {
92
  "epoch": 0.275,
93
- "grad_norm": 0.12842494249343872,
94
  "learning_rate": 3.4936429539683075e-06,
95
- "loss": 10.880451202392578,
96
  "step": 55
97
  },
98
  {
99
  "epoch": 0.3,
100
- "grad_norm": 0.12605500221252441,
101
  "learning_rate": 3.3787068182371314e-06,
102
- "loss": 10.865975952148437,
103
  "step": 60
104
  },
105
  {
106
  "epoch": 0.325,
107
- "grad_norm": 0.12930333614349365,
108
  "learning_rate": 3.254352716947074e-06,
109
- "loss": 10.846955871582031,
110
  "step": 65
111
  },
112
  {
113
  "epoch": 0.35,
114
- "grad_norm": 0.1300428807735443,
115
  "learning_rate": 3.1214301147033453e-06,
116
- "loss": 10.831236267089844,
117
  "step": 70
118
  },
119
  {
120
  "epoch": 0.375,
121
- "grad_norm": 0.13403981924057007,
122
  "learning_rate": 2.9808470076610163e-06,
123
- "loss": 10.815153503417969,
124
  "step": 75
125
  },
126
  {
127
  "epoch": 0.4,
128
- "grad_norm": 0.1414586752653122,
129
  "learning_rate": 2.833563720990581e-06,
130
- "loss": 10.798695373535157,
131
  "step": 80
132
  },
133
  {
134
  "epoch": 0.425,
135
- "grad_norm": 0.1405540555715561,
136
  "learning_rate": 2.680586348883286e-06,
137
- "loss": 10.79158706665039,
138
  "step": 85
139
  },
140
  {
141
  "epoch": 0.45,
142
- "grad_norm": 0.15637332201004028,
143
  "learning_rate": 2.5229598819076393e-06,
144
- "loss": 10.773637390136718,
145
  "step": 90
146
  },
147
  {
148
  "epoch": 0.475,
149
- "grad_norm": 0.15481412410736084,
150
  "learning_rate": 2.36176106866422e-06,
151
- "loss": 10.747270965576172,
152
  "step": 95
153
  },
154
  {
155
  "epoch": 0.5,
156
- "grad_norm": 0.1801270991563797,
157
  "learning_rate": 2.198091060500926e-06,
158
- "loss": 10.725272369384765,
159
  "step": 100
160
  },
161
  {
162
  "epoch": 0.5,
163
- "eval_accuracy": 0.032405372405372404,
164
- "eval_loss": 10.71803092956543,
165
- "eval_runtime": 10.5701,
166
- "eval_samples_per_second": 9.461,
167
- "eval_steps_per_second": 1.608,
168
  "step": 100
169
  }
170
  ],
 
11
  "log_history": [
12
  {
13
  "epoch": 0.025,
14
+ "grad_norm": 0.11061733961105347,
15
  "learning_rate": 1.6e-06,
16
+ "loss": 10.991944885253906,
17
  "step": 5
18
  },
19
  {
20
  "epoch": 0.05,
21
+ "grad_norm": 0.10693935304880142,
22
  "learning_rate": 3.6e-06,
23
+ "loss": 10.94852294921875,
24
  "step": 10
25
  },
26
  {
27
  "epoch": 0.075,
28
+ "grad_norm": 0.10685458779335022,
29
  "learning_rate": 3.995627254437549e-06,
30
+ "loss": 10.901692962646484,
31
  "step": 15
32
  },
33
  {
34
  "epoch": 0.1,
35
+ "grad_norm": 0.11649836599826813,
36
  "learning_rate": 3.9778957412029366e-06,
37
+ "loss": 10.860481262207031,
38
  "step": 20
39
  },
40
  {
41
  "epoch": 0.125,
42
+ "grad_norm": 0.13389746844768524,
43
  "learning_rate": 3.9466531944960116e-06,
44
+ "loss": 10.805369567871093,
45
  "step": 25
46
  },
47
  {
48
  "epoch": 0.15,
49
+ "grad_norm": 0.14611287415027618,
50
  "learning_rate": 3.902113032590307e-06,
51
+ "loss": 10.771086883544921,
52
  "step": 30
53
  },
54
  {
55
  "epoch": 0.175,
56
+ "grad_norm": 0.1613965630531311,
57
  "learning_rate": 3.844579509954769e-06,
58
+ "loss": 10.723188781738282,
59
  "step": 35
60
  },
61
  {
62
  "epoch": 0.2,
63
+ "grad_norm": 0.15956223011016846,
64
  "learning_rate": 3.7744456388872234e-06,
65
+ "loss": 10.692904663085937,
66
  "step": 40
67
  },
68
  {
69
  "epoch": 0.225,
70
+ "grad_norm": 0.19351637363433838,
71
  "learning_rate": 3.6921905048418706e-06,
72
+ "loss": 10.633544921875,
73
  "step": 45
74
  },
75
  {
76
  "epoch": 0.25,
77
+ "grad_norm": 0.19844886660575867,
78
  "learning_rate": 3.598375993789849e-06,
79
+ "loss": 10.588146209716797,
80
  "step": 50
81
  },
82
  {
83
  "epoch": 0.25,
84
+ "eval_accuracy": 0.04886935286935287,
85
+ "eval_loss": 10.55318832397461,
86
+ "eval_runtime": 10.5347,
87
+ "eval_samples_per_second": 9.492,
88
+ "eval_steps_per_second": 1.614,
89
  "step": 50
90
  },
91
  {
92
  "epoch": 0.275,
93
+ "grad_norm": 0.22146697342395782,
94
  "learning_rate": 3.4936429539683075e-06,
95
+ "loss": 10.543079376220703,
96
  "step": 55
97
  },
98
  {
99
  "epoch": 0.3,
100
+ "grad_norm": 0.22231507301330566,
101
  "learning_rate": 3.3787068182371314e-06,
102
+ "loss": 10.494519805908203,
103
  "step": 60
104
  },
105
  {
106
  "epoch": 0.325,
107
+ "grad_norm": 0.2373388409614563,
108
  "learning_rate": 3.254352716947074e-06,
109
+ "loss": 10.443205261230469,
110
  "step": 65
111
  },
112
  {
113
  "epoch": 0.35,
114
+ "grad_norm": 0.24415704607963562,
115
  "learning_rate": 3.1214301147033453e-06,
116
+ "loss": 10.390926361083984,
117
  "step": 70
118
  },
119
  {
120
  "epoch": 0.375,
121
+ "grad_norm": 0.2557603716850281,
122
  "learning_rate": 2.9808470076610163e-06,
123
+ "loss": 10.339105987548828,
124
  "step": 75
125
  },
126
  {
127
  "epoch": 0.4,
128
+ "grad_norm": 0.2784874439239502,
129
  "learning_rate": 2.833563720990581e-06,
130
+ "loss": 10.287814331054687,
131
  "step": 80
132
  },
133
  {
134
  "epoch": 0.425,
135
+ "grad_norm": 0.28266823291778564,
136
  "learning_rate": 2.680586348883286e-06,
137
+ "loss": 10.275057983398437,
138
  "step": 85
139
  },
140
  {
141
  "epoch": 0.45,
142
+ "grad_norm": 0.31389370560646057,
143
  "learning_rate": 2.5229598819076393e-06,
144
+ "loss": 10.22107391357422,
145
  "step": 90
146
  },
147
  {
148
  "epoch": 0.475,
149
+ "grad_norm": 0.3142116963863373,
150
  "learning_rate": 2.36176106866422e-06,
151
+ "loss": 10.138157653808594,
152
  "step": 95
153
  },
154
  {
155
  "epoch": 0.5,
156
+ "grad_norm": 0.37546810507774353,
157
  "learning_rate": 2.198091060500926e-06,
158
+ "loss": 10.064350891113282,
159
  "step": 100
160
  },
161
  {
162
  "epoch": 0.5,
163
+ "eval_accuracy": 0.09671550671550672,
164
+ "eval_loss": 10.042438507080078,
165
+ "eval_runtime": 10.462,
166
+ "eval_samples_per_second": 9.558,
167
+ "eval_steps_per_second": 1.625,
168
  "step": 100
169
  }
170
  ],