CodeIsAbstract commited on
Commit
13dce25
·
verified ·
1 Parent(s): ad759e8

Training in progress, step 6000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9b4228b55bb76fc15ca95505a3ef7ec1042cf993453b65da92e201da75ce6c8d
3
  size 734275920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9bb532cf97353e876d3c07dafda9641b1330c0ef72b36326c6da98516454dc96
3
  size 734275920
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6a5846d1fe27e8749ffc72486781b7806ecb1483360bdd08c503170e6b013162
3
  size 1468687243
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3471614dd25efffa34ced8adea663f06082eaa9ce06ff0ae502a2a2ce49dce6a
3
  size 1468687243
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7de62db38127a19f1b74e4caf4fa2c7f80e3db34a1364e6d48e3845081d03fad
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:15c2f1e87a9fcaf57351b335abc031c00c3adda5167bcf7ec0b18e183f3ee6ae
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d65799baf1d2dfe04d0a138d3793b0b5a501104449fe070829112843354c238d
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:da416040f48730a4d4754962dfc1cf7e84c988b051ed7995fad8903f5b75ce23
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:53e5e0e4abd3c70926da0534c109583a158a9ad33404eba4274e07ea784e7b91
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:96b41d044a0b95f3d68b68c393e43b00c5ca90a1a9bc2f1272a5782f85171f5a
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2ca52ef4191d31f82dcef0a542d277b30f0239b891949e489402bf4e36cbcb3b
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7909009ee6bb6a7388a27f2195f9d99a87ca4bfa7948fb59ebcf32152929a165
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9e477a794f5d1340be38828422d8ce148abb3584a23d53b93905b3bc505514f2
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8793849385d70eb0a4454019ee1dea79e0782bfac3827e1750c91bff2a4e5501
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:86ba7bf6efe64ed70ecf29b7b1a438293fcbe5fba6a7fd4be175bb0a03ccba48
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:aa6203f4502122aa111ee64a6591a834d190fa04fcbd197aea65cf4f041f3e16
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7a8720dd04ab0427bc7dc643de458ddb996dfa7c64ecd5d25e0538512b96776c
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b20e94508e1cbf7f79ffc938a6389a03f2f263a9d7c056df1a5c8567f12db613
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1a79b6ed8c9a33b7a01ec6af6b482537b5b012444212ea60980ce76bcac7ea0d
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f8c8b444dac54dc5ad78a7292762380772bc71cffe63d34cd4898fae9fc500a4
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1a11b408b02a0c1a9e74618915c249c34fbb229118ef1abb560e9617d25e1c52
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d36ffe8038dcfcbb1af52f768d5ded0e7902497baa765ea8664469f5c3a22aec
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.08,
6
  "eval_steps": 1000,
7
- "global_step": 4000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -320,6 +320,162 @@
320
  "eval_samples_per_second": 10.845,
321
  "eval_steps_per_second": 0.542,
322
  "step": 4000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
323
  }
324
  ],
325
  "logging_steps": 100,
@@ -339,7 +495,7 @@
339
  "attributes": {}
340
  }
341
  },
342
- "total_flos": 2.849803075584e+16,
343
  "train_batch_size": 4,
344
  "trial_name": null,
345
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.12,
6
  "eval_steps": 1000,
7
+ "global_step": 6000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
320
  "eval_samples_per_second": 10.845,
321
  "eval_steps_per_second": 0.542,
322
  "step": 4000
323
+ },
324
+ {
325
+ "epoch": 0.082,
326
+ "grad_norm": 160393.71875,
327
+ "learning_rate": 0.000492297867222828,
328
+ "loss": 59.3321435546875,
329
+ "step": 4100
330
+ },
331
+ {
332
+ "epoch": 0.084,
333
+ "grad_norm": 12930.431640625,
334
+ "learning_rate": 0.0004919049934697177,
335
+ "loss": 60.8889404296875,
336
+ "step": 4200
337
+ },
338
+ {
339
+ "epoch": 0.086,
340
+ "grad_norm": 4836.6787109375,
341
+ "learning_rate": 0.0004915025121630086,
342
+ "loss": 61.493935546875,
343
+ "step": 4300
344
+ },
345
+ {
346
+ "epoch": 0.088,
347
+ "grad_norm": 595.083984375,
348
+ "learning_rate": 0.0004910904392877396,
349
+ "loss": 58.8386474609375,
350
+ "step": 4400
351
+ },
352
+ {
353
+ "epoch": 0.09,
354
+ "grad_norm": 12180.2373046875,
355
+ "learning_rate": 0.0004906687912098906,
356
+ "loss": 63.4604150390625,
357
+ "step": 4500
358
+ },
359
+ {
360
+ "epoch": 0.092,
361
+ "grad_norm": 414.4292297363281,
362
+ "learning_rate": 0.0004902375846757322,
363
+ "loss": 59.018310546875,
364
+ "step": 4600
365
+ },
366
+ {
367
+ "epoch": 0.094,
368
+ "grad_norm": 2297.899169921875,
369
+ "learning_rate": 0.0004897968368111611,
370
+ "loss": 57.6424853515625,
371
+ "step": 4700
372
+ },
373
+ {
374
+ "epoch": 0.096,
375
+ "grad_norm": 2821.689697265625,
376
+ "learning_rate": 0.0004893465651210193,
377
+ "loss": 58.3983154296875,
378
+ "step": 4800
379
+ },
380
+ {
381
+ "epoch": 0.098,
382
+ "grad_norm": 2559.970947265625,
383
+ "learning_rate": 0.0004888867874883995,
384
+ "loss": 57.3033935546875,
385
+ "step": 4900
386
+ },
387
+ {
388
+ "epoch": 0.1,
389
+ "grad_norm": 8444.46484375,
390
+ "learning_rate": 0.0004884175221739343,
391
+ "loss": 56.6698193359375,
392
+ "step": 5000
393
+ },
394
+ {
395
+ "epoch": 0.1,
396
+ "eval_loss": 57.39847946166992,
397
+ "eval_runtime": 1.8416,
398
+ "eval_samples_per_second": 10.86,
399
+ "eval_steps_per_second": 0.543,
400
+ "step": 5000
401
+ },
402
+ {
403
+ "epoch": 0.102,
404
+ "grad_norm": 6454.49658203125,
405
+ "learning_rate": 0.0004879387878150716,
406
+ "loss": 56.9357958984375,
407
+ "step": 5100
408
+ },
409
+ {
410
+ "epoch": 0.104,
411
+ "grad_norm": 337538.0,
412
+ "learning_rate": 0.00048745060342533363,
413
+ "loss": 56.982119140625,
414
+ "step": 5200
415
+ },
416
+ {
417
+ "epoch": 0.106,
418
+ "grad_norm": 68615.4609375,
419
+ "learning_rate": 0.0004869529883935625,
420
+ "loss": 56.92587890625,
421
+ "step": 5300
422
+ },
423
+ {
424
+ "epoch": 0.108,
425
+ "grad_norm": 116072.59375,
426
+ "learning_rate": 0.00048644596248314967,
427
+ "loss": 57.190751953125,
428
+ "step": 5400
429
+ },
430
+ {
431
+ "epoch": 0.11,
432
+ "grad_norm": 291359.625,
433
+ "learning_rate": 0.0004859295458312511,
434
+ "loss": 58.6766162109375,
435
+ "step": 5500
436
+ },
437
+ {
438
+ "epoch": 0.112,
439
+ "grad_norm": 89412.578125,
440
+ "learning_rate": 0.0004854037589479878,
441
+ "loss": 58.688193359375,
442
+ "step": 5600
443
+ },
444
+ {
445
+ "epoch": 0.114,
446
+ "grad_norm": 293593.4375,
447
+ "learning_rate": 0.0004848686227156309,
448
+ "loss": 59.063056640625,
449
+ "step": 5700
450
+ },
451
+ {
452
+ "epoch": 0.116,
453
+ "grad_norm": 30479.330078125,
454
+ "learning_rate": 0.0004843241583877724,
455
+ "loss": 58.46580078125,
456
+ "step": 5800
457
+ },
458
+ {
459
+ "epoch": 0.118,
460
+ "grad_norm": 89906.7734375,
461
+ "learning_rate": 0.000483770387588481,
462
+ "loss": 58.2409228515625,
463
+ "step": 5900
464
+ },
465
+ {
466
+ "epoch": 0.12,
467
+ "grad_norm": 630418.5,
468
+ "learning_rate": 0.00048320733231144354,
469
+ "loss": 58.333916015625,
470
+ "step": 6000
471
+ },
472
+ {
473
+ "epoch": 0.12,
474
+ "eval_loss": 59.336669921875,
475
+ "eval_runtime": 1.8507,
476
+ "eval_samples_per_second": 10.807,
477
+ "eval_steps_per_second": 0.54,
478
+ "step": 6000
479
  }
480
  ],
481
  "logging_steps": 100,
 
495
  "attributes": {}
496
  }
497
  },
498
+ "total_flos": 4.274704613376e+16,
499
  "train_batch_size": 4,
500
  "trial_name": null,
501
  "trial_params": null