CodeIsAbstract commited on
Commit
e91c237
·
verified ·
1 Parent(s): 1473f22

Training in progress, step 400, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:66886050e435756ef5a7c49a69ae84363711c5f2bc7b3b5ebe3bf6a9bda5b749
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:44db9200e812bea039f20d9f9d28a6cc957c313c4a850767b225596c0d0b5e4d
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:53f7b73ea97ba098f486f95967fa0a7351e9d606099e9de265782b06a33955a2
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2db0d2136ead3f5f875f36fd5e9c3aabdd558f129a68fea33e8ad5378b962f98
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2f879f5f67a208b766c17cb348fca9d544480e749ad05d8f2acb8625bebbaf11
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:daffcca87be3c0433ad35e4c1c8e3a023512501fd94a8be0b0adb06d69a5a30e
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8a613d380f06e14f7b3ddc807a9d67116b9bcb0317fb8fbbd77c2d40723ed0e8
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:84e96bce876c6af0e48011bb3d66e418a88cfa5db787a6533c7b6b8efb142dbd
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.2,
6
  "eval_steps": 100,
7
- "global_step": 200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -54,6 +54,52 @@
54
  "eval_samples_per_second": 72.664,
55
  "eval_steps_per_second": 4.57,
56
  "step": 200
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
57
  }
58
  ],
59
  "logging_steps": 50,
@@ -73,7 +119,7 @@
73
  "attributes": {}
74
  }
75
  },
76
- "total_flos": 3.34503327301632e+16,
77
  "train_batch_size": 64,
78
  "trial_name": null,
79
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.4,
6
  "eval_steps": 100,
7
+ "global_step": 400,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
54
  "eval_samples_per_second": 72.664,
55
  "eval_steps_per_second": 4.57,
56
  "step": 200
57
+ },
58
+ {
59
+ "epoch": 0.25,
60
+ "grad_norm": 1.5482277870178223,
61
+ "learning_rate": 9.669005017492143e-05,
62
+ "loss": 25.0247,
63
+ "step": 250
64
+ },
65
+ {
66
+ "epoch": 0.3,
67
+ "grad_norm": 3.2663464546203613,
68
+ "learning_rate": 9.26078506448917e-05,
69
+ "loss": 24.2338,
70
+ "step": 300
71
+ },
72
+ {
73
+ "epoch": 0.3,
74
+ "eval_accuracy": 0.1829634890115987,
75
+ "eval_loss": 5.955964088439941,
76
+ "eval_runtime": 12.3645,
77
+ "eval_samples_per_second": 78.45,
78
+ "eval_steps_per_second": 4.933,
79
+ "step": 300
80
+ },
81
+ {
82
+ "epoch": 0.35,
83
+ "grad_norm": 2.2301456928253174,
84
+ "learning_rate": 8.707469186363446e-05,
85
+ "loss": 23.716,
86
+ "step": 350
87
+ },
88
+ {
89
+ "epoch": 0.4,
90
+ "grad_norm": 2.1302545070648193,
91
+ "learning_rate": 8.027899891715287e-05,
92
+ "loss": 23.2729,
93
+ "step": 400
94
+ },
95
+ {
96
+ "epoch": 0.4,
97
+ "eval_accuracy": 0.1966725107382293,
98
+ "eval_loss": 5.7180047035217285,
99
+ "eval_runtime": 13.1479,
100
+ "eval_samples_per_second": 73.776,
101
+ "eval_steps_per_second": 4.64,
102
+ "step": 400
103
  }
104
  ],
105
  "logging_steps": 50,
 
119
  "attributes": {}
120
  }
121
  },
122
+ "total_flos": 6.69006654603264e+16,
123
  "train_batch_size": 64,
124
  "trial_name": null,
125
  "trial_params": null