CodeIsAbstract commited on
Commit
3377730
·
verified ·
1 Parent(s): 8b73486

Training in progress, step 2000, checkpoint

Browse files
last-checkpoint/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bb40413396efb9cb62de4d57c19739b555d21249dac5de20419134364eaccf85
3
+ size 430347
last-checkpoint/pytorch_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1c2c99545af37c1c564bbdb999199ecf2fdcb3e848eb06b5c281c7493b545125
3
+ size 1087359623
last-checkpoint/rng_state_0.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2a2a4adad934312ea001cd84349883cb6c15698df9bb6c090af6bf88181fb06b
3
+ size 14469
last-checkpoint/rng_state_1.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:649e8cc37e14e4911064da2b8cb9ef2ade87f6e78d1185d4b8e5b14453b1c83d
3
+ size 14469
last-checkpoint/rng_state_2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f6403efa0c30bbacae871a7df874ec4bbdb5b528b6d86d3b5609a837c3eb80b9
3
+ size 14469
last-checkpoint/rng_state_3.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fedd614c5b3fb7e22cbb42e3d69fcde0d7a32f2e678c37f7c99bb5a167b975e8
3
+ size 14469
last-checkpoint/rng_state_4.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f72141908b3038f5d527be0a4afe331900fd2cd98b763078b556a00ecca0eca1
3
+ size 14469
last-checkpoint/rng_state_5.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c958787b74a3aefeefe565a04ee3471c563655a4aa29b37f82c1b78d0c1b5a6c
3
+ size 14469
last-checkpoint/rng_state_6.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8a198a4edc5b0a0c0735b02620e42eae859d7b0a942bf7c1a21b08b1885eff98
3
+ size 14469
last-checkpoint/rng_state_7.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1403f69e2c99c4fcaf2999db32c4368d388bb9ac696855db926ab4209f647997
3
+ size 14469
last-checkpoint/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f57060fea471d2fc6e5d97738f7c5e8d6b78209431437727516b20f8639bb057
3
+ size 1465
last-checkpoint/trainer_state.json ADDED
@@ -0,0 +1,192 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 1.0,
6
+ "eval_steps": 1000,
7
+ "global_step": 2000,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": false,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.05,
14
+ "grad_norm": 1.485682487487793,
15
+ "learning_rate": 0.003980291252282261,
16
+ "loss": 73.7283203125,
17
+ "step": 100
18
+ },
19
+ {
20
+ "epoch": 0.1,
21
+ "grad_norm": 1.4347572326660156,
22
+ "learning_rate": 0.003911632443486905,
23
+ "loss": 67.1016357421875,
24
+ "step": 200
25
+ },
26
+ {
27
+ "epoch": 0.15,
28
+ "grad_norm": 1.7708789110183716,
29
+ "learning_rate": 0.003795429623632202,
30
+ "loss": 65.27525390625,
31
+ "step": 300
32
+ },
33
+ {
34
+ "epoch": 0.2,
35
+ "grad_norm": 1.349237322807312,
36
+ "learning_rate": 0.0036345728609271967,
37
+ "loss": 64.15083984375,
38
+ "step": 400
39
+ },
40
+ {
41
+ "epoch": 0.25,
42
+ "grad_norm": 1.3039754629135132,
43
+ "learning_rate": 0.003433062807134769,
44
+ "loss": 63.8335498046875,
45
+ "step": 500
46
+ },
47
+ {
48
+ "epoch": 0.3,
49
+ "grad_norm": 1.2484228610992432,
50
+ "learning_rate": 0.0031959111977790367,
51
+ "loss": 63.246728515625,
52
+ "step": 600
53
+ },
54
+ {
55
+ "epoch": 0.35,
56
+ "grad_norm": 1.4882639646530151,
57
+ "learning_rate": 0.0029290162057940094,
58
+ "loss": 62.8143017578125,
59
+ "step": 700
60
+ },
61
+ {
62
+ "epoch": 0.4,
63
+ "grad_norm": 1.1007429361343384,
64
+ "learning_rate": 0.0026390157486799134,
65
+ "loss": 62.51708984375,
66
+ "step": 800
67
+ },
68
+ {
69
+ "epoch": 0.45,
70
+ "grad_norm": 0.9959006905555725,
71
+ "learning_rate": 0.0023331223975495813,
72
+ "loss": 62.166494140625,
73
+ "step": 900
74
+ },
75
+ {
76
+ "epoch": 0.5,
77
+ "grad_norm": 1.243346929550171,
78
+ "learning_rate": 0.002018943994024803,
79
+ "loss": 62.0639892578125,
80
+ "step": 1000
81
+ },
82
+ {
83
+ "epoch": 0.5,
84
+ "eval_accuracy": 0.15881036168132942,
85
+ "eval_loss": 61.933006286621094,
86
+ "eval_runtime": 21.6587,
87
+ "eval_samples_per_second": 5.771,
88
+ "eval_steps_per_second": 0.185,
89
+ "step": 1000
90
+ },
91
+ {
92
+ "epoch": 0.55,
93
+ "grad_norm": 1.3841041326522827,
94
+ "learning_rate": 0.0017042944364010796,
95
+ "loss": 61.8725634765625,
96
+ "step": 1100
97
+ },
98
+ {
99
+ "epoch": 0.6,
100
+ "grad_norm": 0.9040701389312744,
101
+ "learning_rate": 0.0013969993409983545,
102
+ "loss": 61.6279736328125,
103
+ "step": 1200
104
+ },
105
+ {
106
+ "epoch": 0.65,
107
+ "grad_norm": 0.9380832314491272,
108
+ "learning_rate": 0.0011047014120739685,
109
+ "loss": 61.5527197265625,
110
+ "step": 1300
111
+ },
112
+ {
113
+ "epoch": 0.7,
114
+ "grad_norm": 0.8851712346076965,
115
+ "learning_rate": 0.0008346703609224516,
116
+ "loss": 61.3382958984375,
117
+ "step": 1400
118
+ },
119
+ {
120
+ "epoch": 0.75,
121
+ "grad_norm": 1.1163753271102905,
122
+ "learning_rate": 0.0005936221016443706,
123
+ "loss": 61.1475,
124
+ "step": 1500
125
+ },
126
+ {
127
+ "epoch": 0.8,
128
+ "grad_norm": 0.8482072353363037,
129
+ "learning_rate": 0.0003875517203474137,
130
+ "loss": 61.2019091796875,
131
+ "step": 1600
132
+ },
133
+ {
134
+ "epoch": 0.85,
135
+ "grad_norm": 0.7953028082847595,
136
+ "learning_rate": 0.00022158437198527747,
137
+ "loss": 61.2163671875,
138
+ "step": 1700
139
+ },
140
+ {
141
+ "epoch": 0.9,
142
+ "grad_norm": 0.8152965903282166,
143
+ "learning_rate": 9.984781316353054e-05,
144
+ "loss": 60.9908349609375,
145
+ "step": 1800
146
+ },
147
+ {
148
+ "epoch": 0.95,
149
+ "grad_norm": 0.8798665404319763,
150
+ "learning_rate": 2.536974113572521e-05,
151
+ "loss": 61.030517578125,
152
+ "step": 1900
153
+ },
154
+ {
155
+ "epoch": 1.0,
156
+ "grad_norm": 0.9654077291488647,
157
+ "learning_rate": 2.4922608901079e-09,
158
+ "loss": 61.0981298828125,
159
+ "step": 2000
160
+ },
161
+ {
162
+ "epoch": 1.0,
163
+ "eval_accuracy": 0.1659012707722385,
164
+ "eval_loss": 61.03606033325195,
165
+ "eval_runtime": 6.2006,
166
+ "eval_samples_per_second": 20.159,
167
+ "eval_steps_per_second": 0.645,
168
+ "step": 2000
169
+ }
170
+ ],
171
+ "logging_steps": 100,
172
+ "max_steps": 2000,
173
+ "num_input_tokens_seen": 0,
174
+ "num_train_epochs": 9223372036854775807,
175
+ "save_steps": 2000,
176
+ "stateful_callbacks": {
177
+ "TrainerControl": {
178
+ "args": {
179
+ "should_epoch_stop": false,
180
+ "should_evaluate": false,
181
+ "should_log": false,
182
+ "should_save": true,
183
+ "should_training_stop": true
184
+ },
185
+ "attributes": {}
186
+ }
187
+ },
188
+ "total_flos": 0.0,
189
+ "train_batch_size": 4,
190
+ "trial_name": null,
191
+ "trial_params": null
192
+ }
last-checkpoint/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cf6375b44bdaf1391a7855c6a8372496bc80c4603be4c5f25a7bad8ccbf3839a
3
+ size 5265