raqibcodes commited on
Commit
d3bb775
·
verified ·
1 Parent(s): 11add0f

Training in progress, step 175, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:11f0e2ca13e33a85b3120a4ee713ea0e06007c9189c4a3e7fa7b167a9e3c4f20
3
  size 69839888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b4811d78b9dd78d9aebe7b80047f6217dcfdc6526595cda2721fecd638a5693d
3
  size 69839888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4df2282fccbd13d593ec483631834e137826c7a4f170bcecec0d10f808adb66a
3
  size 139961199
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bb20d043b220c3d5a39ce637102af1f6c75cf597f4b0132882fba06e001f6bd3
3
  size 139961199
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:25a08711b58e82a94c7d430db91c18062e356888f736cdc904051ea42bffee1f
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:170c7959150cf68a6db943e8d36c0961729d2d4eb32726fb3dee9c6cff97c3e7
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:88fff52255c07c1bc4a8c7a336a046e1a58603d0ff7c2a9d4b7c1a94aea45fe9
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0cf6561675164f7b29c0db741edcb28bfc670b31565f97f608dc8729c5ab4465
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -1,10 +1,10 @@
1
  {
2
- "best_global_step": 150,
3
- "best_metric": 0.0021258138585835695,
4
- "best_model_checkpoint": "./c2c_gemma4_e4b_qlora_checkpoint/checkpoint-150",
5
- "epoch": 1.5,
6
  "eval_steps": 25,
7
- "global_step": 150,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -194,6 +194,37 @@
194
  "eval_samples_per_second": 0.716,
195
  "eval_steps_per_second": 0.716,
196
  "step": 150
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
197
  }
198
  ],
199
  "logging_steps": 12,
@@ -213,7 +244,7 @@
213
  "attributes": {}
214
  }
215
  },
216
- "total_flos": 7245552687886080.0,
217
  "train_batch_size": 1,
218
  "trial_name": null,
219
  "trial_params": null
 
1
  {
2
+ "best_global_step": 175,
3
+ "best_metric": 0.0012933596735820174,
4
+ "best_model_checkpoint": "./c2c_gemma4_e4b_qlora_checkpoint/checkpoint-175",
5
+ "epoch": 1.75,
6
  "eval_steps": 25,
7
+ "global_step": 175,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
194
  "eval_samples_per_second": 0.716,
195
  "eval_steps_per_second": 0.716,
196
  "step": 150
197
+ },
198
+ {
199
+ "entropy": 0.9431026068826517,
200
+ "epoch": 1.56,
201
+ "grad_norm": 0.0030975341796875,
202
+ "learning_rate": 9.965635738831616e-05,
203
+ "loss": 0.003447669247786204,
204
+ "mean_token_accuracy": 1.0,
205
+ "num_tokens": 280420.0,
206
+ "step": 156
207
+ },
208
+ {
209
+ "entropy": 0.9445797273268303,
210
+ "epoch": 1.6800000000000002,
211
+ "grad_norm": 0.0017852783203125,
212
+ "learning_rate": 9.140893470790379e-05,
213
+ "loss": 0.0011490107669184606,
214
+ "mean_token_accuracy": 1.0,
215
+ "num_tokens": 302272.0,
216
+ "step": 168
217
+ },
218
+ {
219
+ "epoch": 1.75,
220
+ "eval_entropy": 0.9432125598192215,
221
+ "eval_loss": 0.0012933596735820174,
222
+ "eval_mean_token_accuracy": 0.9997524043917656,
223
+ "eval_num_tokens": 314960.0,
224
+ "eval_runtime": 279.3708,
225
+ "eval_samples_per_second": 0.716,
226
+ "eval_steps_per_second": 0.716,
227
+ "step": 175
228
  }
229
  ],
230
  "logging_steps": 12,
 
244
  "attributes": {}
245
  }
246
  },
247
+ "total_flos": 8478070804454400.0,
248
  "train_batch_size": 1,
249
  "trial_name": null,
250
  "trial_params": null