kiritan commited on
Commit
80bb605
·
verified ·
1 Parent(s): 1c73e82

Training in progress, step 2000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c860a83fb5b9f69118a6d8d4b59a20a11f110515ffcdbc5cd3807e7b1b14b397
3
  size 223144592
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3d272d808c4d78c51ed882e52436b50de43238baa0b3d5032c121b862e8c95ac
3
  size 223144592
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e99dd7e87e0550442caf560a5b4216f0dbb5499430735f63c030308d75e8148e
3
  size 440235130
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:57b32ae9ee23830597b75e7a81d34670c42dbdda022ee958920c8eae3196ea94
3
  size 440235130
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8c05de932ead88280d2522d175713d98e67fbc9cde4f0986f914945e9eeeae9b
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eef8c749264b26d3c435087ceac3890a08b72e1d16e09e5376d4afb6eccc0ec4
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ac8b3aeaa223f49129197d9577ea3c7b67e3ed41a56eef76042b6d22ac2afeb7
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fec4c47fe338ca1d5d1e3d6f4fe9c0141de1fee23f97edeed87e057e0266ffd9
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
- "best_metric": 99.67939651107967,
3
- "best_model_checkpoint": "./iteboshi_temp/checkpoint-1000",
4
- "epoch": 1.6507018992568125,
5
  "eval_steps": 1000,
6
- "global_step": 1000,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -297,6 +297,296 @@
297
  "eval_steps_per_second": 1.1,
298
  "eval_wer": 99.67939651107967,
299
  "step": 1000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
300
  }
301
  ],
302
  "logging_steps": 25,
@@ -316,7 +606,7 @@
316
  "attributes": {}
317
  }
318
  },
319
- "total_flos": 1.94993134239744e+18,
320
  "train_batch_size": 12,
321
  "trial_name": null,
322
  "trial_params": null
 
1
  {
2
+ "best_metric": 97.35030645921735,
3
+ "best_model_checkpoint": "./iteboshi_temp/checkpoint-2000",
4
+ "epoch": 3.300578034682081,
5
  "eval_steps": 1000,
6
+ "global_step": 2000,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
297
  "eval_steps_per_second": 1.1,
298
  "eval_wer": 99.67939651107967,
299
  "step": 1000
300
+ },
301
+ {
302
+ "epoch": 1.6919900908340215,
303
+ "grad_norm": 9.614924430847168,
304
+ "learning_rate": 1.889684210526316e-05,
305
+ "loss": 1.8351,
306
+ "step": 1025
307
+ },
308
+ {
309
+ "epoch": 1.7332782824112303,
310
+ "grad_norm": 9.912728309631348,
311
+ "learning_rate": 1.8844210526315793e-05,
312
+ "loss": 1.7924,
313
+ "step": 1050
314
+ },
315
+ {
316
+ "epoch": 1.7745664739884393,
317
+ "grad_norm": 9.571353912353516,
318
+ "learning_rate": 1.8791578947368423e-05,
319
+ "loss": 1.7655,
320
+ "step": 1075
321
+ },
322
+ {
323
+ "epoch": 1.8158546655656482,
324
+ "grad_norm": 9.284981727600098,
325
+ "learning_rate": 1.8738947368421056e-05,
326
+ "loss": 1.7737,
327
+ "step": 1100
328
+ },
329
+ {
330
+ "epoch": 1.8571428571428572,
331
+ "grad_norm": 10.073586463928223,
332
+ "learning_rate": 1.8686315789473686e-05,
333
+ "loss": 1.7343,
334
+ "step": 1125
335
+ },
336
+ {
337
+ "epoch": 1.8984310487200662,
338
+ "grad_norm": 8.913012504577637,
339
+ "learning_rate": 1.8633684210526316e-05,
340
+ "loss": 1.6881,
341
+ "step": 1150
342
+ },
343
+ {
344
+ "epoch": 1.939719240297275,
345
+ "grad_norm": 9.885432243347168,
346
+ "learning_rate": 1.858105263157895e-05,
347
+ "loss": 1.6909,
348
+ "step": 1175
349
+ },
350
+ {
351
+ "epoch": 1.981007431874484,
352
+ "grad_norm": 9.40440559387207,
353
+ "learning_rate": 1.852842105263158e-05,
354
+ "loss": 1.6276,
355
+ "step": 1200
356
+ },
357
+ {
358
+ "epoch": 2.0214698596201486,
359
+ "grad_norm": 10.476126670837402,
360
+ "learning_rate": 1.8475789473684212e-05,
361
+ "loss": 1.5264,
362
+ "step": 1225
363
+ },
364
+ {
365
+ "epoch": 2.0627580511973576,
366
+ "grad_norm": 8.929309844970703,
367
+ "learning_rate": 1.8423157894736842e-05,
368
+ "loss": 1.4139,
369
+ "step": 1250
370
+ },
371
+ {
372
+ "epoch": 2.1040462427745665,
373
+ "grad_norm": 9.487678527832031,
374
+ "learning_rate": 1.8370526315789476e-05,
375
+ "loss": 1.4765,
376
+ "step": 1275
377
+ },
378
+ {
379
+ "epoch": 2.1453344343517755,
380
+ "grad_norm": 9.428024291992188,
381
+ "learning_rate": 1.831789473684211e-05,
382
+ "loss": 1.4275,
383
+ "step": 1300
384
+ },
385
+ {
386
+ "epoch": 2.1866226259289845,
387
+ "grad_norm": 9.300884246826172,
388
+ "learning_rate": 1.826526315789474e-05,
389
+ "loss": 1.4115,
390
+ "step": 1325
391
+ },
392
+ {
393
+ "epoch": 2.227910817506193,
394
+ "grad_norm": 9.60734748840332,
395
+ "learning_rate": 1.821263157894737e-05,
396
+ "loss": 1.4087,
397
+ "step": 1350
398
+ },
399
+ {
400
+ "epoch": 2.269199009083402,
401
+ "grad_norm": 8.660606384277344,
402
+ "learning_rate": 1.8160000000000002e-05,
403
+ "loss": 1.3658,
404
+ "step": 1375
405
+ },
406
+ {
407
+ "epoch": 2.310487200660611,
408
+ "grad_norm": 9.169739723205566,
409
+ "learning_rate": 1.8107368421052632e-05,
410
+ "loss": 1.3645,
411
+ "step": 1400
412
+ },
413
+ {
414
+ "epoch": 2.35177539223782,
415
+ "grad_norm": 9.20264720916748,
416
+ "learning_rate": 1.8054736842105266e-05,
417
+ "loss": 1.328,
418
+ "step": 1425
419
+ },
420
+ {
421
+ "epoch": 2.393063583815029,
422
+ "grad_norm": 9.625157356262207,
423
+ "learning_rate": 1.8002105263157896e-05,
424
+ "loss": 1.3095,
425
+ "step": 1450
426
+ },
427
+ {
428
+ "epoch": 2.434351775392238,
429
+ "grad_norm": 8.743439674377441,
430
+ "learning_rate": 1.794947368421053e-05,
431
+ "loss": 1.3244,
432
+ "step": 1475
433
+ },
434
+ {
435
+ "epoch": 2.475639966969447,
436
+ "grad_norm": 9.378725051879883,
437
+ "learning_rate": 1.789684210526316e-05,
438
+ "loss": 1.2695,
439
+ "step": 1500
440
+ },
441
+ {
442
+ "epoch": 2.516928158546656,
443
+ "grad_norm": 9.474600791931152,
444
+ "learning_rate": 1.7844210526315792e-05,
445
+ "loss": 1.3238,
446
+ "step": 1525
447
+ },
448
+ {
449
+ "epoch": 2.558216350123865,
450
+ "grad_norm": 8.615851402282715,
451
+ "learning_rate": 1.7791578947368422e-05,
452
+ "loss": 1.2829,
453
+ "step": 1550
454
+ },
455
+ {
456
+ "epoch": 2.5995045417010734,
457
+ "grad_norm": 8.702414512634277,
458
+ "learning_rate": 1.7738947368421052e-05,
459
+ "loss": 1.2914,
460
+ "step": 1575
461
+ },
462
+ {
463
+ "epoch": 2.6407927332782823,
464
+ "grad_norm": 9.374676704406738,
465
+ "learning_rate": 1.7686315789473685e-05,
466
+ "loss": 1.2769,
467
+ "step": 1600
468
+ },
469
+ {
470
+ "epoch": 2.6820809248554913,
471
+ "grad_norm": 8.985688209533691,
472
+ "learning_rate": 1.7633684210526315e-05,
473
+ "loss": 1.25,
474
+ "step": 1625
475
+ },
476
+ {
477
+ "epoch": 2.7233691164327003,
478
+ "grad_norm": 8.864657402038574,
479
+ "learning_rate": 1.758105263157895e-05,
480
+ "loss": 1.2028,
481
+ "step": 1650
482
+ },
483
+ {
484
+ "epoch": 2.7646573080099093,
485
+ "grad_norm": 8.514275550842285,
486
+ "learning_rate": 1.7528421052631582e-05,
487
+ "loss": 1.2477,
488
+ "step": 1675
489
+ },
490
+ {
491
+ "epoch": 2.805945499587118,
492
+ "grad_norm": 9.147972106933594,
493
+ "learning_rate": 1.7475789473684212e-05,
494
+ "loss": 1.2103,
495
+ "step": 1700
496
+ },
497
+ {
498
+ "epoch": 2.847233691164327,
499
+ "grad_norm": 9.296348571777344,
500
+ "learning_rate": 1.7423157894736845e-05,
501
+ "loss": 1.2437,
502
+ "step": 1725
503
+ },
504
+ {
505
+ "epoch": 2.8885218827415358,
506
+ "grad_norm": 9.178862571716309,
507
+ "learning_rate": 1.7370526315789475e-05,
508
+ "loss": 1.1999,
509
+ "step": 1750
510
+ },
511
+ {
512
+ "epoch": 2.9298100743187447,
513
+ "grad_norm": 8.8689603805542,
514
+ "learning_rate": 1.731789473684211e-05,
515
+ "loss": 1.167,
516
+ "step": 1775
517
+ },
518
+ {
519
+ "epoch": 2.9710982658959537,
520
+ "grad_norm": 8.430618286132812,
521
+ "learning_rate": 1.726526315789474e-05,
522
+ "loss": 1.1616,
523
+ "step": 1800
524
+ },
525
+ {
526
+ "epoch": 3.0115606936416186,
527
+ "grad_norm": 8.175687789916992,
528
+ "learning_rate": 1.721263157894737e-05,
529
+ "loss": 1.1536,
530
+ "step": 1825
531
+ },
532
+ {
533
+ "epoch": 3.0528488852188276,
534
+ "grad_norm": 8.549736022949219,
535
+ "learning_rate": 1.7160000000000002e-05,
536
+ "loss": 1.0421,
537
+ "step": 1850
538
+ },
539
+ {
540
+ "epoch": 3.094137076796036,
541
+ "grad_norm": 8.512212753295898,
542
+ "learning_rate": 1.710736842105263e-05,
543
+ "loss": 1.0288,
544
+ "step": 1875
545
+ },
546
+ {
547
+ "epoch": 3.135425268373245,
548
+ "grad_norm": 7.529111385345459,
549
+ "learning_rate": 1.7054736842105265e-05,
550
+ "loss": 0.9965,
551
+ "step": 1900
552
+ },
553
+ {
554
+ "epoch": 3.176713459950454,
555
+ "grad_norm": 8.706369400024414,
556
+ "learning_rate": 1.7002105263157895e-05,
557
+ "loss": 1.0262,
558
+ "step": 1925
559
+ },
560
+ {
561
+ "epoch": 3.218001651527663,
562
+ "grad_norm": 8.718097686767578,
563
+ "learning_rate": 1.6949473684210528e-05,
564
+ "loss": 1.0232,
565
+ "step": 1950
566
+ },
567
+ {
568
+ "epoch": 3.259289843104872,
569
+ "grad_norm": 9.124544143676758,
570
+ "learning_rate": 1.689684210526316e-05,
571
+ "loss": 1.0104,
572
+ "step": 1975
573
+ },
574
+ {
575
+ "epoch": 3.300578034682081,
576
+ "grad_norm": 7.853435039520264,
577
+ "learning_rate": 1.684421052631579e-05,
578
+ "loss": 0.9948,
579
+ "step": 2000
580
+ },
581
+ {
582
+ "epoch": 3.300578034682081,
583
+ "eval_cer": 59.421319913335545,
584
+ "eval_loss": 1.2762852907180786,
585
+ "eval_runtime": 738.0908,
586
+ "eval_samples_per_second": 14.336,
587
+ "eval_steps_per_second": 1.195,
588
+ "eval_wer": 97.35030645921735,
589
+ "step": 2000
590
  }
591
  ],
592
  "logging_steps": 25,
 
606
  "attributes": {}
607
  }
608
  },
609
+ "total_flos": 3.89856186335232e+18,
610
  "train_batch_size": 12,
611
  "trial_name": null,
612
  "trial_params": null