madeofajala commited on
Commit
5c8296d
·
verified ·
1 Parent(s): 1fc78ce

Training in progress, step 3300, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:85faf2af15805afc584779874d529ad28534d9b80913768169c519e815904657
3
  size 20814808
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fc19303a0059a9bd3cef9383344c1970c866cef910a80be81fce23bbaec6c4a7
3
  size 20814808
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:da57250fb14c3d817e7a3315a9cd0640494a511f590814ab4432059cd139bdb9
3
  size 21506325
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7697fc27b0dc2afe5eeb1609893e8b1a0638f7c1a60aa51849627255d02d8a05
3
  size 21506325
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:df1807568898653709b82a2b64cebe0bfb59e7a412ebb63d00d62bebc5eeeb44
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e1868f5b3985b17bc9517aa4adad17b968c0ca25cc39f9c06ba06f94a13b6df3
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.9925746325200788,
6
  "eval_steps": 300,
7
- "global_step": 3275,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1318,6 +1318,16 @@
1318
  "mean_token_accuracy": 0.9430344653129578,
1319
  "num_tokens": 5355357.0,
1320
  "step": 3275
 
 
 
 
 
 
 
 
 
 
1321
  }
1322
  ],
1323
  "logging_steps": 25,
@@ -1332,12 +1342,12 @@
1332
  "should_evaluate": false,
1333
  "should_log": false,
1334
  "should_save": true,
1335
- "should_training_stop": false
1336
  },
1337
  "attributes": {}
1338
  }
1339
  },
1340
- "total_flos": 7.151728825938739e+16,
1341
  "train_batch_size": 2,
1342
  "trial_name": null,
1343
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 1.0,
6
  "eval_steps": 300,
7
+ "global_step": 3300,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1318
  "mean_token_accuracy": 0.9430344653129578,
1319
  "num_tokens": 5355357.0,
1320
  "step": 3275
1321
+ },
1322
+ {
1323
+ "entropy": 0.15088965865422269,
1324
+ "epoch": 1.0,
1325
+ "grad_norm": 0.4375,
1326
+ "learning_rate": 0.0002,
1327
+ "loss": 0.1564,
1328
+ "mean_token_accuracy": 0.9488337550844465,
1329
+ "num_tokens": 5394110.0,
1330
+ "step": 3300
1331
  }
1332
  ],
1333
  "logging_steps": 25,
 
1342
  "should_evaluate": false,
1343
  "should_log": false,
1344
  "should_save": true,
1345
+ "should_training_stop": true
1346
  },
1347
  "attributes": {}
1348
  }
1349
  },
1350
+ "total_flos": 7.203637122878669e+16,
1351
  "train_batch_size": 2,
1352
  "trial_name": null,
1353
  "trial_params": null