VedaX-0.7B-Base / trainer_state.json
VedaX-Labs's picture
Upload 10 files
59e27e6 verified
Raw
History Blame Contribute Delete
5.05 kB
{
"best_global_step": 500,
"best_metric": 0.04982214793562889,
"best_model_checkpoint": "C:\\Users\\rumpa\\OneDrive\\Desktop\\VedaX_AI_finetune\\VedaX_AI\\checkpoint-500",
"epoch": 0.05387205387205387,
"eval_steps": 500,
"global_step": 500,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.0026936026936026937,
"grad_norm": 1.841068983078003,
"learning_rate": 1.997414350355527e-05,
"loss": 2.8212152099609376,
"step": 25
},
{
"epoch": 0.0053872053872053875,
"grad_norm": 0.7459769248962402,
"learning_rate": 1.9947209653092007e-05,
"loss": 0.36805130004882813,
"step": 50
},
{
"epoch": 0.00808080808080808,
"grad_norm": 0.3883790969848633,
"learning_rate": 1.9920275802628747e-05,
"loss": 0.08150457382202149,
"step": 75
},
{
"epoch": 0.010774410774410775,
"grad_norm": 0.24909840524196625,
"learning_rate": 1.9893341952165484e-05,
"loss": 0.06001442432403564,
"step": 100
},
{
"epoch": 0.013468013468013467,
"grad_norm": 0.23919348418712616,
"learning_rate": 1.986640810170222e-05,
"loss": 0.05564042568206787,
"step": 125
},
{
"epoch": 0.01616161616161616,
"grad_norm": 0.21730053424835205,
"learning_rate": 1.983947425123896e-05,
"loss": 0.053155436515808105,
"step": 150
},
{
"epoch": 0.018855218855218854,
"grad_norm": 0.345330148935318,
"learning_rate": 1.98125404007757e-05,
"loss": 0.05441353797912598,
"step": 175
},
{
"epoch": 0.02154882154882155,
"grad_norm": 0.2703924775123596,
"learning_rate": 1.9785606550312436e-05,
"loss": 0.05095250129699707,
"step": 200
},
{
"epoch": 0.024242424242424242,
"grad_norm": 0.2786615788936615,
"learning_rate": 1.9758672699849173e-05,
"loss": 0.050014009475708006,
"step": 225
},
{
"epoch": 0.026936026936026935,
"grad_norm": 0.22033053636550903,
"learning_rate": 1.973173884938591e-05,
"loss": 0.05086544513702393,
"step": 250
},
{
"epoch": 0.02962962962962963,
"grad_norm": 0.28292080760002136,
"learning_rate": 1.9704804998922647e-05,
"loss": 0.05024292469024658,
"step": 275
},
{
"epoch": 0.03232323232323232,
"grad_norm": 0.15764561295509338,
"learning_rate": 1.9677871148459384e-05,
"loss": 0.04976308822631836,
"step": 300
},
{
"epoch": 0.035016835016835016,
"grad_norm": 0.26158377528190613,
"learning_rate": 1.965093729799612e-05,
"loss": 0.050297212600708005,
"step": 325
},
{
"epoch": 0.03771043771043771,
"grad_norm": 0.32958391308784485,
"learning_rate": 1.962400344753286e-05,
"loss": 0.049454569816589355,
"step": 350
},
{
"epoch": 0.04040404040404041,
"grad_norm": 0.18668946623802185,
"learning_rate": 1.95970695970696e-05,
"loss": 0.05047787666320801,
"step": 375
},
{
"epoch": 0.0430976430976431,
"grad_norm": 0.2564663887023926,
"learning_rate": 1.9570135746606336e-05,
"loss": 0.0484027624130249,
"step": 400
},
{
"epoch": 0.04579124579124579,
"grad_norm": 0.1699737310409546,
"learning_rate": 1.9543201896143073e-05,
"loss": 0.048322863578796386,
"step": 425
},
{
"epoch": 0.048484848484848485,
"grad_norm": 0.19934280216693878,
"learning_rate": 1.951626804567981e-05,
"loss": 0.04795527935028076,
"step": 450
},
{
"epoch": 0.05117845117845118,
"grad_norm": 0.18609412014484406,
"learning_rate": 1.948933419521655e-05,
"loss": 0.049452505111694335,
"step": 475
},
{
"epoch": 0.05387205387205387,
"grad_norm": 0.15728649497032166,
"learning_rate": 1.9462400344753287e-05,
"loss": 0.048011641502380374,
"step": 500
},
{
"epoch": 0.05387205387205387,
"eval_loss": 0.04982214793562889,
"eval_runtime": 144.2281,
"eval_samples_per_second": 10.4,
"eval_steps_per_second": 10.4,
"step": 500
}
],
"logging_steps": 25,
"max_steps": 18564,
"num_input_tokens_seen": 0,
"num_train_epochs": 2,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 392321487886848.0,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}