rovdetection commited on
Commit
2485828
·
verified ·
1 Parent(s): bcb208f

Training in progress, step 1000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ad6aeef38ed358f8048c115091424d7e45cd8e98e79fdc324106be4910f58c95
3
  size 4523108832
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6b3bd15ad5f6e246b6e337225788c69c49a21df6eeb262d2b1b1897cf5cf6146
3
  size 4523108832
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b2d76978d088fcb730ac283219f0e32e79d7d521d55adaf8f7006b69f804ac5c
3
  size 2911851147
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9ecf8f056195b0098a8274b3035e24672255ff4c10ed92a0717c3ca84f2c7426
3
  size 2911851147
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:098b29492211804ab324a36f37466821d948280bb74fce4ba895c03f13ecd878
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a8e2011629d8bed3ef560fa11175cac55684c4e12a72634bb24abf767b6c7399
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f77569c2e850b04af982cc8c1389f1430851448915c593b69e5da36ce05b71d7
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:14ae2a2128444abab378aa06c09a61a84665f758fcc19fc46f5789b0bc1b5665
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ff3b68c80bb6e1918e0fcbbfb26fa72db3f12e89d05c20abd2bd696373cfaadc
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:92235ff997c3a5622fba1c344122ccffab641b57d7443401a19efb448a747750
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -1,10 +1,10 @@
1
  {
2
- "best_global_step": 500,
3
- "best_metric": 1.041561245918274,
4
- "best_model_checkpoint": "./sft-out/checkpoint-500",
5
- "epoch": 0.8840267418089397,
6
  "eval_steps": 500,
7
- "global_step": 500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -86,6 +86,84 @@
86
  "eval_samples_per_second": 18.184,
87
  "eval_steps_per_second": 2.291,
88
  "step": 500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
89
  }
90
  ],
91
  "logging_steps": 50,
@@ -105,7 +183,7 @@
105
  "attributes": {}
106
  }
107
  },
108
- "total_flos": 1.9429307822186496e+16,
109
  "train_batch_size": 1,
110
  "trial_name": null,
111
  "trial_params": null
 
1
  {
2
+ "best_global_step": 1000,
3
+ "best_metric": 0.988082230091095,
4
+ "best_model_checkpoint": "./sft-out/checkpoint-1000",
5
+ "epoch": 1.7673352118901597,
6
  "eval_steps": 500,
7
+ "global_step": 1000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
86
  "eval_samples_per_second": 18.184,
87
  "eval_steps_per_second": 2.291,
88
  "step": 500
89
+ },
90
+ {
91
+ "epoch": 0.9724294159898337,
92
+ "grad_norm": 2.2783753871917725,
93
+ "learning_rate": 1.884018956444297e-05,
94
+ "loss": 1.068794708251953,
95
+ "step": 550
96
+ },
97
+ {
98
+ "epoch": 1.060113818443008,
99
+ "grad_norm": 2.0879323482513428,
100
+ "learning_rate": 1.85741517679431e-05,
101
+ "loss": 0.8203845977783203,
102
+ "step": 600
103
+ },
104
+ {
105
+ "epoch": 1.148516492623902,
106
+ "grad_norm": 2.290803909301758,
107
+ "learning_rate": 1.828296450700078e-05,
108
+ "loss": 0.7069022369384765,
109
+ "step": 650
110
+ },
111
+ {
112
+ "epoch": 1.2369191668047959,
113
+ "grad_norm": 2.6409525871276855,
114
+ "learning_rate": 1.7967481884023257e-05,
115
+ "loss": 0.707320556640625,
116
+ "step": 700
117
+ },
118
+ {
119
+ "epoch": 1.3253218409856897,
120
+ "grad_norm": 2.363614559173584,
121
+ "learning_rate": 1.7628629263900676e-05,
122
+ "loss": 0.7089004516601562,
123
+ "step": 750
124
+ },
125
+ {
126
+ "epoch": 1.4137245151665838,
127
+ "grad_norm": 2.3598053455352783,
128
+ "learning_rate": 1.7267400559751396e-05,
129
+ "loss": 0.709156265258789,
130
+ "step": 800
131
+ },
132
+ {
133
+ "epoch": 1.502127189347478,
134
+ "grad_norm": 2.0364632606506348,
135
+ "learning_rate": 1.6884855317603584e-05,
136
+ "loss": 0.6911470031738282,
137
+ "step": 850
138
+ },
139
+ {
140
+ "epoch": 1.5905298635283718,
141
+ "grad_norm": 2.0562117099761963,
142
+ "learning_rate": 1.648211560856415e-05,
143
+ "loss": 0.6989401245117187,
144
+ "step": 900
145
+ },
146
+ {
147
+ "epoch": 1.6789325377092656,
148
+ "grad_norm": 2.351238965988159,
149
+ "learning_rate": 1.6060362737590845e-05,
150
+ "loss": 0.7044075775146484,
151
+ "step": 950
152
+ },
153
+ {
154
+ "epoch": 1.7673352118901597,
155
+ "grad_norm": 2.1121199131011963,
156
+ "learning_rate": 1.5620833778521306e-05,
157
+ "loss": 0.7032404327392578,
158
+ "step": 1000
159
+ },
160
+ {
161
+ "epoch": 1.7673352118901597,
162
+ "eval_loss": 0.988082230091095,
163
+ "eval_runtime": 27.3518,
164
+ "eval_samples_per_second": 18.28,
165
+ "eval_steps_per_second": 2.303,
166
+ "step": 1000
167
  }
168
  ],
169
  "logging_steps": 50,
 
183
  "attributes": {}
184
  }
185
  },
186
+ "total_flos": 3.876769443016704e+16,
187
  "train_batch_size": 1,
188
  "trial_name": null,
189
  "trial_params": null