CodeIsAbstract commited on
Commit
f15e2c8
·
verified ·
1 Parent(s): 737a64d

Training in progress, step 32000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ac7b6e8e0e6c9527838b5b71a909559d9e98612b291cc816fa041bd4e932bc2f
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4abcfe19109678f34d3ff11a8d3ce7a037cf1f3d94b5bd2c29643daf81f1deb6
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3e73756a742f5a52e3fb1f593643683cf7bad589d676e48aa89861e9865dadfc
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4ed11d32ed680b24cb759e32cd2f16ac63d1901c6c0f38c4ba936bb380de271e
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7d60e00b2d8189621a65a613490321cabbb70d9187223365d4f127c3fc0b2584
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c7ce608857b08326fa7d341f30263d7dc70be1ec0abf4c06d041f987c5ac0d05
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:02a3be2e7ecb88f30f5b2994e37142ffea2663c1e78cc9ead86da93ad8248f79
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eb48a6869c32358f585d366d73d143bd6791297810d0a877b008b19b1b4c14d7
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.6,
6
  "eval_steps": 1000,
7
- "global_step": 30000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2378,6 +2378,164 @@
2378
  "eval_samples_per_second": 305.046,
2379
  "eval_steps_per_second": 19.173,
2380
  "step": 30000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2381
  }
2382
  ],
2383
  "logging_steps": 100,
@@ -2397,7 +2555,7 @@
2397
  "attributes": {}
2398
  }
2399
  },
2400
- "total_flos": 1.1759882600448e+18,
2401
  "train_batch_size": 120,
2402
  "trial_name": null,
2403
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.64,
6
  "eval_steps": 1000,
7
+ "global_step": 32000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2378
  "eval_samples_per_second": 305.046,
2379
  "eval_steps_per_second": 19.173,
2380
  "step": 30000
2381
+ },
2382
+ {
2383
+ "epoch": 0.602,
2384
+ "grad_norm": 0.14657410979270935,
2385
+ "learning_rate": 0.0003443232098442347,
2386
+ "loss": 3.3668572998046873,
2387
+ "step": 30100
2388
+ },
2389
+ {
2390
+ "epoch": 0.604,
2391
+ "grad_norm": 0.138699010014534,
2392
+ "learning_rate": 0.0003413319007702201,
2393
+ "loss": 3.3502297973632813,
2394
+ "step": 30200
2395
+ },
2396
+ {
2397
+ "epoch": 0.606,
2398
+ "grad_norm": 0.15832597017288208,
2399
+ "learning_rate": 0.0003383468933944299,
2400
+ "loss": 3.355211181640625,
2401
+ "step": 30300
2402
+ },
2403
+ {
2404
+ "epoch": 0.608,
2405
+ "grad_norm": 0.15029780566692352,
2406
+ "learning_rate": 0.0003353683062700939,
2407
+ "loss": 3.348283386230469,
2408
+ "step": 30400
2409
+ },
2410
+ {
2411
+ "epoch": 0.61,
2412
+ "grad_norm": 0.17595185339450836,
2413
+ "learning_rate": 0.0003323962576954539,
2414
+ "loss": 3.3513107299804688,
2415
+ "step": 30500
2416
+ },
2417
+ {
2418
+ "epoch": 0.612,
2419
+ "grad_norm": 0.1754045933485031,
2420
+ "learning_rate": 0.00032943086570906576,
2421
+ "loss": 3.3677444458007812,
2422
+ "step": 30600
2423
+ },
2424
+ {
2425
+ "epoch": 0.614,
2426
+ "grad_norm": 0.14994262158870697,
2427
+ "learning_rate": 0.00032647224808511,
2428
+ "loss": 3.3526513671875,
2429
+ "step": 30700
2430
+ },
2431
+ {
2432
+ "epoch": 0.616,
2433
+ "grad_norm": 0.14404474198818207,
2434
+ "learning_rate": 0.00032352052232871543,
2435
+ "loss": 3.31786865234375,
2436
+ "step": 30800
2437
+ },
2438
+ {
2439
+ "epoch": 0.618,
2440
+ "grad_norm": 0.14154170453548431,
2441
+ "learning_rate": 0.0003205758056712919,
2442
+ "loss": 3.341680908203125,
2443
+ "step": 30900
2444
+ },
2445
+ {
2446
+ "epoch": 0.62,
2447
+ "grad_norm": 0.13272085785865784,
2448
+ "learning_rate": 0.0003176382150658742,
2449
+ "loss": 3.344057922363281,
2450
+ "step": 31000
2451
+ },
2452
+ {
2453
+ "epoch": 0.62,
2454
+ "eval_accuracy": 0.3808566004369608,
2455
+ "eval_loss": 3.3054208755493164,
2456
+ "eval_runtime": 6.7684,
2457
+ "eval_samples_per_second": 286.774,
2458
+ "eval_steps_per_second": 18.025,
2459
+ "step": 31000
2460
+ },
2461
+ {
2462
+ "epoch": 0.622,
2463
+ "grad_norm": 0.14439266920089722,
2464
+ "learning_rate": 0.00031470786718247704,
2465
+ "loss": 3.336947021484375,
2466
+ "step": 31100
2467
+ },
2468
+ {
2469
+ "epoch": 0.624,
2470
+ "grad_norm": 0.1481838971376419,
2471
+ "learning_rate": 0.000311784878403462,
2472
+ "loss": 3.3591497802734374,
2473
+ "step": 31200
2474
+ },
2475
+ {
2476
+ "epoch": 0.626,
2477
+ "grad_norm": 0.13957072794437408,
2478
+ "learning_rate": 0.00030886936481891447,
2479
+ "loss": 3.333016357421875,
2480
+ "step": 31300
2481
+ },
2482
+ {
2483
+ "epoch": 0.628,
2484
+ "grad_norm": 0.1470046043395996,
2485
+ "learning_rate": 0.0003059614422220331,
2486
+ "loss": 3.349130859375,
2487
+ "step": 31400
2488
+ },
2489
+ {
2490
+ "epoch": 0.63,
2491
+ "grad_norm": 0.13627968728542328,
2492
+ "learning_rate": 0.00030306122610453183,
2493
+ "loss": 3.3334747314453126,
2494
+ "step": 31500
2495
+ },
2496
+ {
2497
+ "epoch": 0.632,
2498
+ "grad_norm": 0.27877408266067505,
2499
+ "learning_rate": 0.00030016883165205166,
2500
+ "loss": 3.3201373291015623,
2501
+ "step": 31600
2502
+ },
2503
+ {
2504
+ "epoch": 0.634,
2505
+ "grad_norm": 0.14691542088985443,
2506
+ "learning_rate": 0.00029728437373958684,
2507
+ "loss": 3.3420004272460937,
2508
+ "step": 31700
2509
+ },
2510
+ {
2511
+ "epoch": 0.636,
2512
+ "grad_norm": 0.1655166745185852,
2513
+ "learning_rate": 0.0002944079669269224,
2514
+ "loss": 3.32868408203125,
2515
+ "step": 31800
2516
+ },
2517
+ {
2518
+ "epoch": 0.638,
2519
+ "grad_norm": 0.16263002157211304,
2520
+ "learning_rate": 0.0002915397254540836,
2521
+ "loss": 3.327845153808594,
2522
+ "step": 31900
2523
+ },
2524
+ {
2525
+ "epoch": 0.64,
2526
+ "grad_norm": 0.13551194965839386,
2527
+ "learning_rate": 0.00028867976323679957,
2528
+ "loss": 3.326793518066406,
2529
+ "step": 32000
2530
+ },
2531
+ {
2532
+ "epoch": 0.64,
2533
+ "eval_accuracy": 0.38135970019690457,
2534
+ "eval_loss": 3.302846908569336,
2535
+ "eval_runtime": 6.3825,
2536
+ "eval_samples_per_second": 304.112,
2537
+ "eval_steps_per_second": 19.115,
2538
+ "step": 32000
2539
  }
2540
  ],
2541
  "logging_steps": 100,
 
2555
  "attributes": {}
2556
  }
2557
  },
2558
+ "total_flos": 1.25438747738112e+18,
2559
  "train_batch_size": 120,
2560
  "trial_name": null,
2561
  "trial_params": null