madeofajala commited on
Commit
ce47c09
·
verified ·
1 Parent(s): a4f1121

Training in progress, step 2550, checkpoint

Browse files
last-checkpoint/adapter_config.json CHANGED
@@ -29,13 +29,13 @@
29
  "rank_pattern": {},
30
  "revision": null,
31
  "target_modules": [
32
- "o_proj",
33
  "up_proj",
34
- "gate_proj",
35
- "k_proj",
36
  "v_proj",
37
  "down_proj",
38
- "q_proj"
 
 
39
  ],
40
  "target_parameters": null,
41
  "task_type": "CAUSAL_LM",
 
29
  "rank_pattern": {},
30
  "revision": null,
31
  "target_modules": [
32
+ "q_proj",
33
  "up_proj",
 
 
34
  "v_proj",
35
  "down_proj",
36
+ "o_proj",
37
+ "k_proj",
38
+ "gate_proj"
39
  ],
40
  "target_parameters": null,
41
  "task_type": "CAUSAL_LM",
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:378c742a2ed47583532794cae1fd2de4a6760ea2e1b8121476178b01d980da3e
3
  size 108113968
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dcb38d1a14c924041fe5a80f020327deb0212523052b88a9869515999fb8b0d4
3
  size 108113968
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a898a183983728143a712f6bc3eb4cff78c82a63ea213dd37b8d81025cbe4b5c
3
  size 57081771
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f9295ab1b0a55031b1697cf0d73be84f706f3df721cf031ba0ba34ebee613008
3
  size 57081771
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9e3a9e09f9c50cf219bb4c5d3609a4ab4f8f9e31801cf50cb4cced79e186978e
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6fb274dfbe503274d6dd4f2521e2755230b18e2c0f55f4480cf566471adb1def
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.7651515151515151,
6
  "eval_steps": 300,
7
- "global_step": 2525,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1018,10 +1018,20 @@
1018
  "mean_token_accuracy": 0.9421089863777161,
1019
  "num_tokens": 123713.0,
1020
  "step": 2525
 
 
 
 
 
 
 
 
 
 
1021
  }
1022
  ],
1023
  "logging_steps": 25,
1024
- "max_steps": 3300,
1025
  "num_input_tokens_seen": 0,
1026
  "num_train_epochs": 1,
1027
  "save_steps": 25,
@@ -1037,7 +1047,7 @@
1037
  "attributes": {}
1038
  }
1039
  },
1040
- "total_flos": 2.4410656586072064e+17,
1041
  "train_batch_size": 4,
1042
  "trial_name": null,
1043
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.5796771993634917,
6
  "eval_steps": 300,
7
+ "global_step": 2550,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1018
  "mean_token_accuracy": 0.9421089863777161,
1019
  "num_tokens": 123713.0,
1020
  "step": 2525
1021
+ },
1022
+ {
1023
+ "entropy": 0.16171722620725631,
1024
+ "epoch": 0.5796771993634917,
1025
+ "grad_norm": 0.33440494537353516,
1026
+ "learning_rate": 0.0002,
1027
+ "loss": 0.15710799217224122,
1028
+ "mean_token_accuracy": 0.9462839317321777,
1029
+ "num_tokens": 31337.0,
1030
+ "step": 2550
1031
  }
1032
  ],
1033
  "logging_steps": 25,
1034
+ "max_steps": 4399,
1035
  "num_input_tokens_seen": 0,
1036
  "num_train_epochs": 1,
1037
  "save_steps": 25,
 
1047
  "attributes": {}
1048
  }
1049
  },
1050
+ "total_flos": 2.4591659063217254e+17,
1051
  "train_batch_size": 4,
1052
  "trial_name": null,
1053
  "trial_params": null
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:960f819df65000b60022084047fdd37ac0488cbb84d6a7a54a670acd7a6a59f6
3
  size 5649
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9f44b81cef0de28756796a239128e158d8d837ede815af7c9b8b55df63dc7629
3
  size 5649