CodeIsAbstract commited on
Commit
a4ac3b1
·
verified ·
1 Parent(s): 0e0a790

Training in progress, step 20, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2e5eb4a883937e96c950093ab4d6b86746b8c7e18ce6915bc47da02a55ffd429
3
  size 847599616
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fca60095f86e48a89c115bed3de4d6e6d51ec2903783fab00ce19205817f38c5
3
  size 847599616
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f5317de9449b47a23fdbfcc31bffea58037950a2d4a4c2f2a29c8e87618fd6a5
3
  size 350603
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:67a4bb936e346a4b958bd78c0646fdd71067c03febe3739e01fb6f283bec2fc9
3
  size 350603
last-checkpoint/trainer_state.json CHANGED
@@ -11,38 +11,38 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.025,
14
- "grad_norm": 0.12107392400503159,
15
  "learning_rate": 1.6e-06,
16
- "loss": 10.970177459716798,
17
  "step": 5
18
  },
19
  {
20
  "epoch": 0.05,
21
- "grad_norm": 0.11410612612962723,
22
  "learning_rate": 3.6e-06,
23
- "loss": 10.963343811035156,
24
  "step": 10
25
  },
26
  {
27
  "epoch": 0.075,
28
- "grad_norm": 0.10789366811513901,
29
  "learning_rate": 3.995627254437549e-06,
30
- "loss": 10.965518951416016,
31
  "step": 15
32
  },
33
  {
34
  "epoch": 0.1,
35
- "grad_norm": 0.10765355825424194,
36
  "learning_rate": 3.9778957412029366e-06,
37
- "loss": 10.960431671142578,
38
  "step": 20
39
  },
40
  {
41
  "epoch": 0.1,
42
- "eval_accuracy": 9.4993894993895e-05,
43
- "eval_loss": 10.96417236328125,
44
- "eval_runtime": 91.763,
45
- "eval_samples_per_second": 10.898,
46
  "eval_steps_per_second": 1.82,
47
  "step": 20
48
  }
 
11
  "log_history": [
12
  {
13
  "epoch": 0.025,
14
+ "grad_norm": 0.11081302911043167,
15
  "learning_rate": 1.6e-06,
16
+ "loss": 11.017050170898438,
17
  "step": 5
18
  },
19
  {
20
  "epoch": 0.05,
21
+ "grad_norm": 0.1046549528837204,
22
  "learning_rate": 3.6e-06,
23
+ "loss": 11.006002807617188,
24
  "step": 10
25
  },
26
  {
27
  "epoch": 0.075,
28
+ "grad_norm": 0.1001407653093338,
29
  "learning_rate": 3.995627254437549e-06,
30
+ "loss": 10.999064636230468,
31
  "step": 15
32
  },
33
  {
34
  "epoch": 0.1,
35
+ "grad_norm": 0.10080653429031372,
36
  "learning_rate": 3.9778957412029366e-06,
37
+ "loss": 10.995188903808593,
38
  "step": 20
39
  },
40
  {
41
  "epoch": 0.1,
42
+ "eval_accuracy": 1.7094017094017095e-05,
43
+ "eval_loss": 10.983185768127441,
44
+ "eval_runtime": 91.7452,
45
+ "eval_samples_per_second": 10.9,
46
  "eval_steps_per_second": 1.82,
47
  "step": 20
48
  }