madhuHuggingface commited on
Commit
f8f733c
·
verified ·
1 Parent(s): 3e3247e

Training in progress, step 4900, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6c13770aef864f19b029c11df6b654920ade8eab1e57dae1504f856efaf53fc1
3
  size 121537408
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e1087e4a0b3f3c50858aaa17da0f7f26270ae7243622ee70c50ee678c76b041c
3
  size 121537408
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1f8542a2057f1cb46cf9dc9cc12a07f3b4bbfdab7fda142686a284ee3af3ab75
3
  size 62000725
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:78ce1e7d2e218c6ffb0ee36f88013c777ef07192679ce1147a7f52fc07a3b389
3
  size 62000725
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a61d015e9bc0166a3b7a73dda1a07b26c2b7c781f683ff19269df751d3b0698c
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5c7d9ebd4185557c61c160a6357a2c3da0202e4806e8f3f679ac8bee0416df92
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 1.6472758472758473,
6
  "eval_steps": 500,
7
- "global_step": 4800,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -3368,6 +3368,76 @@
3368
  "learning_rate": 8.500080194428753e-05,
3369
  "loss": 0.0016,
3370
  "step": 4800
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3371
  }
3372
  ],
3373
  "logging_steps": 10,
@@ -3387,7 +3457,7 @@
3387
  "attributes": {}
3388
  }
3389
  },
3390
- "total_flos": 1.820330184342605e+16,
3391
  "train_batch_size": 2,
3392
  "trial_name": null,
3393
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 1.6815958815958816,
6
  "eval_steps": 500,
7
+ "global_step": 4900,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
3368
  "learning_rate": 8.500080194428753e-05,
3369
  "loss": 0.0016,
3370
  "step": 4800
3371
+ },
3372
+ {
3373
+ "epoch": 1.6507078507078508,
3374
+ "grad_norm": 0.007407157216221094,
3375
+ "learning_rate": 8.46447830112171e-05,
3376
+ "loss": 0.0018,
3377
+ "step": 4810
3378
+ },
3379
+ {
3380
+ "epoch": 1.6541398541398542,
3381
+ "grad_norm": 0.07495221495628357,
3382
+ "learning_rate": 8.428896329362068e-05,
3383
+ "loss": 0.0024,
3384
+ "step": 4820
3385
+ },
3386
+ {
3387
+ "epoch": 1.6575718575718574,
3388
+ "grad_norm": 0.021885816007852554,
3389
+ "learning_rate": 8.393334740783122e-05,
3390
+ "loss": 0.0026,
3391
+ "step": 4830
3392
+ },
3393
+ {
3394
+ "epoch": 1.661003861003861,
3395
+ "grad_norm": 0.002211750252172351,
3396
+ "learning_rate": 8.357793996753712e-05,
3397
+ "loss": 0.0018,
3398
+ "step": 4840
3399
+ },
3400
+ {
3401
+ "epoch": 1.6644358644358643,
3402
+ "grad_norm": 0.060478098690509796,
3403
+ "learning_rate": 8.322274558372257e-05,
3404
+ "loss": 0.004,
3405
+ "step": 4850
3406
+ },
3407
+ {
3408
+ "epoch": 1.667867867867868,
3409
+ "grad_norm": 0.04104934260249138,
3410
+ "learning_rate": 8.286776886460746e-05,
3411
+ "loss": 0.0028,
3412
+ "step": 4860
3413
+ },
3414
+ {
3415
+ "epoch": 1.6712998712998712,
3416
+ "grad_norm": 0.0045183151960372925,
3417
+ "learning_rate": 8.251301441558789e-05,
3418
+ "loss": 0.0059,
3419
+ "step": 4870
3420
+ },
3421
+ {
3422
+ "epoch": 1.6747318747318747,
3423
+ "grad_norm": 0.4653756320476532,
3424
+ "learning_rate": 8.215848683917616e-05,
3425
+ "loss": 0.0026,
3426
+ "step": 4880
3427
+ },
3428
+ {
3429
+ "epoch": 1.6781638781638781,
3430
+ "grad_norm": 0.048206135630607605,
3431
+ "learning_rate": 8.180419073494125e-05,
3432
+ "loss": 0.0025,
3433
+ "step": 4890
3434
+ },
3435
+ {
3436
+ "epoch": 1.6815958815958816,
3437
+ "grad_norm": 0.0014776921598240733,
3438
+ "learning_rate": 8.145013069944896e-05,
3439
+ "loss": 0.0021,
3440
+ "step": 4900
3441
  }
3442
  ],
3443
  "logging_steps": 10,
 
3457
  "attributes": {}
3458
  }
3459
  },
3460
+ "total_flos": 1.858281325933133e+16,
3461
  "train_batch_size": 2,
3462
  "trial_name": null,
3463
  "trial_params": null