raqibcodes commited on
Commit
1d6b9d2
·
verified ·
1 Parent(s): 2211214

Training in progress, step 50, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4999667755eca122c92f4656a93ce065f1fd4ae04ce109a7ec9defdd3301c4f8
3
  size 69839888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b2f52210b2cf87e0ccd1dd365c144ff12b98d432d03d1182d05b04c027944b56
3
  size 69839888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9d564a6a8aab42f3fb1121a32e0900d1b535aea150ee2ceac579cb24a4eca395
3
  size 139961199
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dcc65ca6791256127638e354db5c80aaa7c3c229250f9fb2c6316327c20cad73
3
  size 139961199
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c30f65627209559594157ae614631b8d56698fb24be27669b8bb4cd12c736775
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4373c49731cc6142e94702f0818c32829d15507599cd9c38571955e14b0c337e
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:25de4c86a54b92b0e6bd0af2e34bf608ee7461eef78f96730d623152d2081ecb
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4d06a4f6b5bf353f960c09cd8473cd505cec37e78e2c01e0308679bf4aaaa00e
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -1,10 +1,10 @@
1
  {
2
- "best_global_step": 25,
3
- "best_metric": 0.02136003039777279,
4
- "best_model_checkpoint": "./c2c_gemma4_e4b_qlora_checkpoint/checkpoint-25",
5
- "epoch": 0.25,
6
  "eval_steps": 25,
7
- "global_step": 25,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -39,6 +39,37 @@
39
  "eval_samples_per_second": 0.715,
40
  "eval_steps_per_second": 0.715,
41
  "step": 25
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
42
  }
43
  ],
44
  "logging_steps": 12,
@@ -58,7 +89,7 @@
58
  "attributes": {}
59
  }
60
  },
61
- "total_flos": 1208049719434560.0,
62
  "train_batch_size": 1,
63
  "trial_name": null,
64
  "trial_params": null
 
1
  {
2
+ "best_global_step": 50,
3
+ "best_metric": 0.00427626259624958,
4
+ "best_model_checkpoint": "./c2c_gemma4_e4b_qlora_checkpoint/checkpoint-50",
5
+ "epoch": 0.5,
6
  "eval_steps": 25,
7
+ "global_step": 50,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
39
  "eval_samples_per_second": 0.715,
40
  "eval_steps_per_second": 0.715,
41
  "step": 25
42
+ },
43
+ {
44
+ "entropy": 1.0095942709594965,
45
+ "epoch": 0.36,
46
+ "grad_norm": 0.1298828125,
47
+ "learning_rate": 0.00018213058419243986,
48
+ "loss": 0.14251227180163065,
49
+ "mean_token_accuracy": 0.9945974703878164,
50
+ "num_tokens": 64401.0,
51
+ "step": 36
52
+ },
53
+ {
54
+ "entropy": 1.0027960228423278,
55
+ "epoch": 0.48,
56
+ "grad_norm": 0.09423828125,
57
+ "learning_rate": 0.0001738831615120275,
58
+ "loss": 0.06836261848608653,
59
+ "mean_token_accuracy": 0.9983609306315581,
60
+ "num_tokens": 86300.0,
61
+ "step": 48
62
+ },
63
+ {
64
+ "epoch": 0.5,
65
+ "eval_entropy": 0.9924230831861496,
66
+ "eval_loss": 0.00427626259624958,
67
+ "eval_mean_token_accuracy": 0.9989724820852279,
68
+ "eval_num_tokens": 89986.0,
69
+ "eval_runtime": 278.956,
70
+ "eval_samples_per_second": 0.717,
71
+ "eval_steps_per_second": 0.717,
72
+ "step": 50
73
  }
74
  ],
75
  "logging_steps": 12,
 
89
  "attributes": {}
90
  }
91
  },
92
+ "total_flos": 2422236726599040.0,
93
  "train_batch_size": 1,
94
  "trial_name": null,
95
  "trial_params": null