CodeIsAbstract commited on
Commit
0a42e21
·
verified ·
1 Parent(s): 16951af

Training in progress, step 600, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:44db9200e812bea039f20d9f9d28a6cc957c313c4a850767b225596c0d0b5e4d
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:28169bb183740976fb1e1012281faf4daaa901f0809070b5431b167a9aed9330
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2db0d2136ead3f5f875f36fd5e9c3aabdd558f129a68fea33e8ad5378b962f98
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5bcc09d2987fc871b0f3d765698df5a7ddb67b242d4bf83d5e4d4194e97c62f1
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:daffcca87be3c0433ad35e4c1c8e3a023512501fd94a8be0b0adb06d69a5a30e
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ecde0592a54835c0791d413df0b84cfe61c8375c6c53371a3a262f4a720ce7ee
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:84e96bce876c6af0e48011bb3d66e418a88cfa5db787a6533c7b6b8efb142dbd
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bff1f3af0c59bc6ebaa29224579d5dfcd2f5ed0cc9dfc92a0a059a14be64f7bc
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.4,
6
  "eval_steps": 100,
7
- "global_step": 400,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -100,6 +100,52 @@
100
  "eval_samples_per_second": 73.776,
101
  "eval_steps_per_second": 4.64,
102
  "step": 400
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
103
  }
104
  ],
105
  "logging_steps": 50,
@@ -119,7 +165,7 @@
119
  "attributes": {}
120
  }
121
  },
122
- "total_flos": 6.69006654603264e+16,
123
  "train_batch_size": 64,
124
  "trial_name": null,
125
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.6,
6
  "eval_steps": 100,
7
+ "global_step": 600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
100
  "eval_samples_per_second": 73.776,
101
  "eval_steps_per_second": 4.64,
102
  "step": 400
103
+ },
104
+ {
105
+ "epoch": 0.45,
106
+ "grad_norm": 2.3908166885375977,
107
+ "learning_rate": 7.24521909782041e-05,
108
+ "loss": 22.8638,
109
+ "step": 450
110
+ },
111
+ {
112
+ "epoch": 0.5,
113
+ "grad_norm": 2.105929136276245,
114
+ "learning_rate": 6.386080060392754e-05,
115
+ "loss": 22.5651,
116
+ "step": 500
117
+ },
118
+ {
119
+ "epoch": 0.5,
120
+ "eval_accuracy": 0.20586229645784704,
121
+ "eval_loss": 5.55440616607666,
122
+ "eval_runtime": 13.4924,
123
+ "eval_samples_per_second": 71.892,
124
+ "eval_steps_per_second": 4.521,
125
+ "step": 500
126
+ },
127
+ {
128
+ "epoch": 0.55,
129
+ "grad_norm": 2.184967041015625,
130
+ "learning_rate": 5.479739728928219e-05,
131
+ "loss": 22.2741,
132
+ "step": 550
133
+ },
134
+ {
135
+ "epoch": 0.6,
136
+ "grad_norm": 2.2348268032073975,
137
+ "learning_rate": 4.5570624363794226e-05,
138
+ "loss": 22.0473,
139
+ "step": 600
140
+ },
141
+ {
142
+ "epoch": 0.6,
143
+ "eval_accuracy": 0.21187852939218293,
144
+ "eval_loss": 5.435864448547363,
145
+ "eval_runtime": 12.3748,
146
+ "eval_samples_per_second": 78.385,
147
+ "eval_steps_per_second": 4.929,
148
+ "step": 600
149
  }
150
  ],
151
  "logging_steps": 50,
 
165
  "attributes": {}
166
  }
167
  },
168
+ "total_flos": 1.003509981904896e+17,
169
  "train_batch_size": 64,
170
  "trial_name": null,
171
  "trial_params": null