CodeIsAbstract commited on
Commit
eacaaa2
·
verified ·
1 Parent(s): 407823f

Training in progress, step 1000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b8845500a4dc1993ae580f65cd736c2ce8c1bff15668b71002335c45a51a47e8
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:642456a6bc85176e3566b9427d8beb7115a0fafcd5b27b0b59eb42535308fbde
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4954cdf965a3940fc6de41f864ead162d174b26dbc7b9885233f859011033e6f
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2518af05d741accc73f1a505524ce249c33241014d55c9eea9f6b9eb85600988
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:11940f1313899a11d3e47a2d43f508134dd8e03ac7613f4eca32c754da2d1839
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c5be6e3d57dd80805cee595c88f58b1448944d9ae8da6e7751e52aec6eab1ba6
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:16279ec4c3e65a1b1efa54fea9187732ebc8942013041dc30c624760de297a84
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b58b3e84a4f76126054e827b2a46c40bd977fea6cc5ec216bb0dc3e5192c23af
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.9,
6
  "eval_steps": 500,
7
- "global_step": 900,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -143,6 +143,29 @@
143
  "learning_rate": 3.443464128710239e-05,
144
  "loss": 16.0619,
145
  "step": 900
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
146
  }
147
  ],
148
  "logging_steps": 50,
@@ -157,12 +180,12 @@
157
  "should_evaluate": false,
158
  "should_log": false,
159
  "should_save": true,
160
- "should_training_stop": false
161
  },
162
  "attributes": {}
163
  }
164
  },
165
- "total_flos": 1.505264972857344e+17,
166
  "train_batch_size": 64,
167
  "trial_name": null,
168
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 1.0,
6
  "eval_steps": 500,
7
+ "global_step": 1000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
143
  "learning_rate": 3.443464128710239e-05,
144
  "loss": 16.0619,
145
  "step": 900
146
+ },
147
+ {
148
+ "epoch": 0.95,
149
+ "grad_norm": 0.6535219550132751,
150
+ "learning_rate": 8.856374635655695e-06,
151
+ "loss": 16.0151,
152
+ "step": 950
153
+ },
154
+ {
155
+ "epoch": 1.0,
156
+ "grad_norm": 0.7225023508071899,
157
+ "learning_rate": 3.4150841404789744e-09,
158
+ "loss": 16.0566,
159
+ "step": 1000
160
+ },
161
+ {
162
+ "epoch": 1.0,
163
+ "eval_accuracy": 0.32063273339130327,
164
+ "eval_loss": 3.969637870788574,
165
+ "eval_runtime": 9.1804,
166
+ "eval_samples_per_second": 105.66,
167
+ "eval_steps_per_second": 6.645,
168
+ "step": 1000
169
  }
170
  ],
171
  "logging_steps": 50,
 
180
  "should_evaluate": false,
181
  "should_log": false,
182
  "should_save": true,
183
+ "should_training_stop": true
184
  },
185
  "attributes": {}
186
  }
187
  },
188
+ "total_flos": 1.67251663650816e+17,
189
  "train_batch_size": 64,
190
  "trial_name": null,
191
  "trial_params": null