CodeIsAbstract commited on
Commit
58254de
·
verified ·
1 Parent(s): 91641c5

Training in progress, step 30000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:101dd85060f77fe6ab87ec8a72c51f4493aee0cd0bb19e810811a805ed98aad4
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ac7b6e8e0e6c9527838b5b71a909559d9e98612b291cc816fa041bd4e932bc2f
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:54ba4d04e29776946b001a6f6920fd7c30f03e9863aff3ad6b551f23caea0b60
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3e73756a742f5a52e3fb1f593643683cf7bad589d676e48aa89861e9865dadfc
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ee44ef43900d0428f565a6ab0fb42d678379f6436ece4e4a6239f399d8eada96
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7d60e00b2d8189621a65a613490321cabbb70d9187223365d4f127c3fc0b2584
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bf7545cbdf140980776a20b719262192e6070ee27cb1c9a27c369c20b3d411df
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:02a3be2e7ecb88f30f5b2994e37142ffea2663c1e78cc9ead86da93ad8248f79
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.56,
6
  "eval_steps": 1000,
7
- "global_step": 28000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2220,6 +2220,164 @@
2220
  "eval_samples_per_second": 292.932,
2221
  "eval_steps_per_second": 18.412,
2222
  "step": 28000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2223
  }
2224
  ],
2225
  "logging_steps": 100,
@@ -2239,7 +2397,7 @@
2239
  "attributes": {}
2240
  }
2241
  },
2242
- "total_flos": 1.09758904270848e+18,
2243
  "train_batch_size": 120,
2244
  "trial_name": null,
2245
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.6,
6
  "eval_steps": 1000,
7
+ "global_step": 30000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2220
  "eval_samples_per_second": 292.932,
2221
  "eval_steps_per_second": 18.412,
2222
  "step": 28000
2223
+ },
2224
+ {
2225
+ "epoch": 0.562,
2226
+ "grad_norm": 0.14532077312469482,
2227
+ "learning_rate": 0.0004052881194593424,
2228
+ "loss": 3.372833251953125,
2229
+ "step": 28100
2230
+ },
2231
+ {
2232
+ "epoch": 0.564,
2233
+ "grad_norm": 0.13675428926944733,
2234
+ "learning_rate": 0.0004021960232718194,
2235
+ "loss": 3.3700936889648436,
2236
+ "step": 28200
2237
+ },
2238
+ {
2239
+ "epoch": 0.566,
2240
+ "grad_norm": 0.15135474503040314,
2241
+ "learning_rate": 0.00039910781148922364,
2242
+ "loss": 3.3734249877929687,
2243
+ "step": 28300
2244
+ },
2245
+ {
2246
+ "epoch": 0.568,
2247
+ "grad_norm": 0.1487869918346405,
2248
+ "learning_rate": 0.00039602360676367474,
2249
+ "loss": 3.3548593139648437,
2250
+ "step": 28400
2251
+ },
2252
+ {
2253
+ "epoch": 0.57,
2254
+ "grad_norm": 0.12845182418823242,
2255
+ "learning_rate": 0.00039294353158814783,
2256
+ "loss": 3.3592022705078124,
2257
+ "step": 28500
2258
+ },
2259
+ {
2260
+ "epoch": 0.572,
2261
+ "grad_norm": 0.15905176103115082,
2262
+ "learning_rate": 0.0003898677082916068,
2263
+ "loss": 3.375569152832031,
2264
+ "step": 28600
2265
+ },
2266
+ {
2267
+ "epoch": 0.574,
2268
+ "grad_norm": 0.1350318342447281,
2269
+ "learning_rate": 0.0003867962590341477,
2270
+ "loss": 3.363053894042969,
2271
+ "step": 28700
2272
+ },
2273
+ {
2274
+ "epoch": 0.576,
2275
+ "grad_norm": 0.15002867579460144,
2276
+ "learning_rate": 0.00038372930580214595,
2277
+ "loss": 3.3725015258789064,
2278
+ "step": 28800
2279
+ },
2280
+ {
2281
+ "epoch": 0.578,
2282
+ "grad_norm": 0.14523345232009888,
2283
+ "learning_rate": 0.000380666970403412,
2284
+ "loss": 3.375093688964844,
2285
+ "step": 28900
2286
+ },
2287
+ {
2288
+ "epoch": 0.58,
2289
+ "grad_norm": 0.1521645337343216,
2290
+ "learning_rate": 0.000377609374462353,
2291
+ "loss": 3.350813903808594,
2292
+ "step": 29000
2293
+ },
2294
+ {
2295
+ "epoch": 0.58,
2296
+ "eval_accuracy": 0.3793452847252259,
2297
+ "eval_loss": 3.322114944458008,
2298
+ "eval_runtime": 6.9103,
2299
+ "eval_samples_per_second": 280.884,
2300
+ "eval_steps_per_second": 17.655,
2301
+ "step": 29000
2302
+ },
2303
+ {
2304
+ "epoch": 0.582,
2305
+ "grad_norm": 0.14382348954677582,
2306
+ "learning_rate": 0.00037455663941514335,
2307
+ "loss": 3.3781756591796874,
2308
+ "step": 29100
2309
+ },
2310
+ {
2311
+ "epoch": 0.584,
2312
+ "grad_norm": 0.1457841694355011,
2313
+ "learning_rate": 0.00037150888650490057,
2314
+ "loss": 3.3773403930664063,
2315
+ "step": 29200
2316
+ },
2317
+ {
2318
+ "epoch": 0.586,
2319
+ "grad_norm": 0.13661234080791473,
2320
+ "learning_rate": 0.0003684662367768703,
2321
+ "loss": 3.3766348266601565,
2322
+ "step": 29300
2323
+ },
2324
+ {
2325
+ "epoch": 0.588,
2326
+ "grad_norm": 0.16734236478805542,
2327
+ "learning_rate": 0.00036542881107361983,
2328
+ "loss": 3.3571905517578124,
2329
+ "step": 29400
2330
+ },
2331
+ {
2332
+ "epoch": 0.59,
2333
+ "grad_norm": 0.14157910645008087,
2334
+ "learning_rate": 0.00036239673003023745,
2335
+ "loss": 3.366629638671875,
2336
+ "step": 29500
2337
+ },
2338
+ {
2339
+ "epoch": 0.592,
2340
+ "grad_norm": 0.13826872408390045,
2341
+ "learning_rate": 0.0003593701140695413,
2342
+ "loss": 3.3789642333984373,
2343
+ "step": 29600
2344
+ },
2345
+ {
2346
+ "epoch": 0.594,
2347
+ "grad_norm": 0.1601548194885254,
2348
+ "learning_rate": 0.00035634908339729813,
2349
+ "loss": 3.37031005859375,
2350
+ "step": 29700
2351
+ },
2352
+ {
2353
+ "epoch": 0.596,
2354
+ "grad_norm": 0.13469478487968445,
2355
+ "learning_rate": 0.00035333375799744704,
2356
+ "loss": 3.3534210205078123,
2357
+ "step": 29800
2358
+ },
2359
+ {
2360
+ "epoch": 0.598,
2361
+ "grad_norm": 0.12652449309825897,
2362
+ "learning_rate": 0.00035032425762733554,
2363
+ "loss": 3.34040283203125,
2364
+ "step": 29900
2365
+ },
2366
+ {
2367
+ "epoch": 0.6,
2368
+ "grad_norm": 0.18712250888347626,
2369
+ "learning_rate": 0.0003473207018129635,
2370
+ "loss": 3.356614685058594,
2371
+ "step": 30000
2372
+ },
2373
+ {
2374
+ "epoch": 0.6,
2375
+ "eval_accuracy": 0.3802063011480555,
2376
+ "eval_loss": 3.314772367477417,
2377
+ "eval_runtime": 6.363,
2378
+ "eval_samples_per_second": 305.046,
2379
+ "eval_steps_per_second": 19.173,
2380
+ "step": 30000
2381
  }
2382
  ],
2383
  "logging_steps": 100,
 
2397
  "attributes": {}
2398
  }
2399
  },
2400
+ "total_flos": 1.1759882600448e+18,
2401
  "train_batch_size": 120,
2402
  "trial_name": null,
2403
  "trial_params": null