CodeIsAbstract commited on
Commit
4b1d345
·
verified ·
1 Parent(s): 1c519a4

Training in progress, step 6000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7ebb38a924e948ebe7405d836c0326774ec3e803d6c16572a9234fd3887d27e5
3
  size 1600779
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4c06f8bdb3470386515b85743f5f433fba520ab14f1b589e6f665e843ff544c6
3
  size 1600779
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d391ca485a0db15ae264d0aafa0da7edc68c0f696b30fd294f28e503817b0a28
3
  size 621149155
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:47079a3c6b76024ae75741e62abdf81d9e0ad0f98cded6e05c983e0a05682bad
3
  size 621149155
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:851f1a1d183e5897767447c672bcf8362d9ae2a4736cbb50b4e208139d171c19
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d2a3ee2b9270eba0a2311f907ea53e1bec546e16e72a9c2fda6872be77c59c2f
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a7fa122671d1ef4d22de37e01b14987409ecd975ea71c5bf79191d6ce98cf8a3
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1b4ff89f8eafc39f075c9216838e6c051cd47ba4912a50dce51d8507d2b54554
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0836353fc5b4695271900d529f17e5c9bcc8f6b960c6c8ad2842449359842642
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:949e92b373ef785d8153838a856e7e94fcdbceccf0df94fac6b1428fba0aa52c
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d85458cac81ee63dbddcc21880d9d5d65d2999a521682ced8bb92cd38533d72b
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06a6a2db4cb588d8ce4130d4e05d07a9c55b5589d5cb162b1a3d954475ff88d3
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5b61eff2b0a5194e1f7cf6b678cdf994e3194534b171d100afdd85158f083dbf
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7088f74e466de020dfa05daf938b6e89670196138b802849a37053644f6497b8
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9e09f720647dea58d44bd8a7064bc267a53d19c77ef833401671fb959ec0bee3
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0e94d95b4f9bc5f7112993c3b1fffc2e8c2ce3186f5ea05083591485e3d86055
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4fc874467b18473948fa002d73f9ced910bc93328612c168e89eefd3ff25af66
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:abab3cbb093f08eba5cc70d5ebf9d591bdbafc0cbf9d3444496ddf74c29c9cf8
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e642abe2620703f2d27e2aefa3f96715c8ae63a82165f9e63e498f488c2c681d
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5972654a558bb8bded698917db31ad4f8203a51ad38c3abf88508a55e8574b82
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:76cf98a8cde877f1e49a9e4b884c4dce3d51bb4035429d464e74acc35ddca024
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eb2a89a85e1c8ac7ee9b5678d857a4d99090faa37a24489f44839e0801dfe16a
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.08,
6
  "eval_steps": 1000,
7
- "global_step": 4000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -324,6 +324,164 @@
324
  "eval_samples_per_second": 39.147,
325
  "eval_steps_per_second": 0.626,
326
  "step": 4000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
327
  }
328
  ],
329
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.12,
6
  "eval_steps": 1000,
7
+ "global_step": 6000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
324
  "eval_samples_per_second": 39.147,
325
  "eval_steps_per_second": 0.626,
326
  "step": 4000
327
+ },
328
+ {
329
+ "epoch": 0.082,
330
+ "grad_norm": 17.625349044799805,
331
+ "learning_rate": 0.0019944130517800893,
332
+ "loss": 36.6922,
333
+ "step": 4100
334
+ },
335
+ {
336
+ "epoch": 0.084,
337
+ "grad_norm": 2.45036244392395,
338
+ "learning_rate": 0.001993693153591457,
339
+ "loss": 36.347,
340
+ "step": 4200
341
+ },
342
+ {
343
+ "epoch": 0.086,
344
+ "grad_norm": 3.7300970554351807,
345
+ "learning_rate": 0.0019929297880451674,
346
+ "loss": 35.9307,
347
+ "step": 4300
348
+ },
349
+ {
350
+ "epoch": 0.088,
351
+ "grad_norm": 1.9578310251235962,
352
+ "learning_rate": 0.0019921229885333023,
353
+ "loss": 35.2814,
354
+ "step": 4400
355
+ },
356
+ {
357
+ "epoch": 0.09,
358
+ "grad_norm": 2.398409843444824,
359
+ "learning_rate": 0.001991272790347886,
360
+ "loss": 35.4533,
361
+ "step": 4500
362
+ },
363
+ {
364
+ "epoch": 0.092,
365
+ "grad_norm": 9.658867835998535,
366
+ "learning_rate": 0.0019903792306793415,
367
+ "loss": 36.307,
368
+ "step": 4600
369
+ },
370
+ {
371
+ "epoch": 0.094,
372
+ "grad_norm": 1.590939998626709,
373
+ "learning_rate": 0.001989442348614863,
374
+ "loss": 36.1738,
375
+ "step": 4700
376
+ },
377
+ {
378
+ "epoch": 0.096,
379
+ "grad_norm": 1.983271837234497,
380
+ "learning_rate": 0.0019884621851367075,
381
+ "loss": 35.8103,
382
+ "step": 4800
383
+ },
384
+ {
385
+ "epoch": 0.098,
386
+ "grad_norm": 2.5749049186706543,
387
+ "learning_rate": 0.001987438783120401,
388
+ "loss": 35.6467,
389
+ "step": 4900
390
+ },
391
+ {
392
+ "epoch": 0.1,
393
+ "grad_norm": 2.067998170852661,
394
+ "learning_rate": 0.001986372187332862,
395
+ "loss": 35.323,
396
+ "step": 5000
397
+ },
398
+ {
399
+ "epoch": 0.1,
400
+ "eval_accuracy": 0.26483953033268104,
401
+ "eval_loss": 35.489601135253906,
402
+ "eval_runtime": 3.2889,
403
+ "eval_samples_per_second": 38.007,
404
+ "eval_steps_per_second": 0.608,
405
+ "step": 5000
406
+ },
407
+ {
408
+ "epoch": 0.102,
409
+ "grad_norm": 1.8774443864822388,
410
+ "learning_rate": 0.0019852624444304467,
411
+ "loss": 34.9978,
412
+ "step": 5100
413
+ },
414
+ {
415
+ "epoch": 0.104,
416
+ "grad_norm": 2.0262346267700195,
417
+ "learning_rate": 0.0019841096029569044,
418
+ "loss": 34.6675,
419
+ "step": 5200
420
+ },
421
+ {
422
+ "epoch": 0.106,
423
+ "grad_norm": 1.776671051979065,
424
+ "learning_rate": 0.0019829137133412556,
425
+ "loss": 35.4506,
426
+ "step": 5300
427
+ },
428
+ {
429
+ "epoch": 0.108,
430
+ "grad_norm": 1.751766324043274,
431
+ "learning_rate": 0.001981674827895587,
432
+ "loss": 35.3834,
433
+ "step": 5400
434
+ },
435
+ {
436
+ "epoch": 0.11,
437
+ "grad_norm": 2.024419069290161,
438
+ "learning_rate": 0.00198039300081276,
439
+ "loss": 35.2425,
440
+ "step": 5500
441
+ },
442
+ {
443
+ "epoch": 0.112,
444
+ "grad_norm": 1.8287601470947266,
445
+ "learning_rate": 0.0019790682881640448,
446
+ "loss": 35.3261,
447
+ "step": 5600
448
+ },
449
+ {
450
+ "epoch": 0.114,
451
+ "grad_norm": 1.6531343460083008,
452
+ "learning_rate": 0.001977700747896664,
453
+ "loss": 34.8158,
454
+ "step": 5700
455
+ },
456
+ {
457
+ "epoch": 0.116,
458
+ "grad_norm": 2.180936098098755,
459
+ "learning_rate": 0.001976290439831259,
460
+ "loss": 34.4037,
461
+ "step": 5800
462
+ },
463
+ {
464
+ "epoch": 0.118,
465
+ "grad_norm": 1.750649094581604,
466
+ "learning_rate": 0.0019748374256592736,
467
+ "loss": 33.958,
468
+ "step": 5900
469
+ },
470
+ {
471
+ "epoch": 0.12,
472
+ "grad_norm": 1.9409301280975342,
473
+ "learning_rate": 0.0019733417689402543,
474
+ "loss": 34.2086,
475
+ "step": 6000
476
+ },
477
+ {
478
+ "epoch": 0.12,
479
+ "eval_accuracy": 0.2734716242661448,
480
+ "eval_loss": 34.69782257080078,
481
+ "eval_runtime": 3.2919,
482
+ "eval_samples_per_second": 37.972,
483
+ "eval_steps_per_second": 0.608,
484
+ "step": 6000
485
  }
486
  ],
487
  "logging_steps": 100,