CodeIsAbstract commited on
Commit
9347050
·
verified ·
1 Parent(s): 8ed86ef

Training in progress, step 28000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:766f39cf2aa03b06ce0544465f23f42d796935dadd6619740980847fa379f296
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:101dd85060f77fe6ab87ec8a72c51f4493aee0cd0bb19e810811a805ed98aad4
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ad971091ff30bc232ebf11b75c27a84c1325701e13bc4a6246fcc2d26e336bb4
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:54ba4d04e29776946b001a6f6920fd7c30f03e9863aff3ad6b551f23caea0b60
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:232768e06875e03d77948d18a567c6062bc64ee278f8d41937981a714e6a2d53
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ee44ef43900d0428f565a6ab0fb42d678379f6436ece4e4a6239f399d8eada96
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2c5343924a6852b4d8168211c95036da6881053e86461e5c21fd663c6b1f4a7b
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bf7545cbdf140980776a20b719262192e6070ee27cb1c9a27c369c20b3d411df
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.52,
6
  "eval_steps": 1000,
7
- "global_step": 26000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2062,6 +2062,164 @@
2062
  "eval_samples_per_second": 297.843,
2063
  "eval_steps_per_second": 18.721,
2064
  "step": 26000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2065
  }
2066
  ],
2067
  "logging_steps": 100,
@@ -2081,7 +2239,7 @@
2081
  "attributes": {}
2082
  }
2083
  },
2084
- "total_flos": 1.01918982537216e+18,
2085
  "train_batch_size": 120,
2086
  "trial_name": null,
2087
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.56,
6
  "eval_steps": 1000,
7
+ "global_step": 28000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2062
  "eval_samples_per_second": 297.843,
2063
  "eval_steps_per_second": 18.721,
2064
  "step": 26000
2065
+ },
2066
+ {
2067
+ "epoch": 0.522,
2068
+ "grad_norm": 0.14482583105564117,
2069
+ "learning_rate": 0.0004677556825558636,
2070
+ "loss": 3.380513000488281,
2071
+ "step": 26100
2072
+ },
2073
+ {
2074
+ "epoch": 0.524,
2075
+ "grad_norm": 0.1740926057100296,
2076
+ "learning_rate": 0.0004646118569767831,
2077
+ "loss": 3.3777206420898436,
2078
+ "step": 26200
2079
+ },
2080
+ {
2081
+ "epoch": 0.526,
2082
+ "grad_norm": 0.13794536888599396,
2083
+ "learning_rate": 0.00046146943688121875,
2084
+ "loss": 3.3788577270507814,
2085
+ "step": 26300
2086
+ },
2087
+ {
2088
+ "epoch": 0.528,
2089
+ "grad_norm": 0.14884917438030243,
2090
+ "learning_rate": 0.00045832854707424044,
2091
+ "loss": 3.372579345703125,
2092
+ "step": 26400
2093
+ },
2094
+ {
2095
+ "epoch": 0.53,
2096
+ "grad_norm": 0.14392748475074768,
2097
+ "learning_rate": 0.0004551893123001399,
2098
+ "loss": 3.369126281738281,
2099
+ "step": 26500
2100
+ },
2101
+ {
2102
+ "epoch": 0.532,
2103
+ "grad_norm": 0.1415095329284668,
2104
+ "learning_rate": 0.0004520518572374776,
2105
+ "loss": 3.4073892211914063,
2106
+ "step": 26600
2107
+ },
2108
+ {
2109
+ "epoch": 0.534,
2110
+ "grad_norm": 0.17241106927394867,
2111
+ "learning_rate": 0.00044891630649413096,
2112
+ "loss": 3.3727621459960937,
2113
+ "step": 26700
2114
+ },
2115
+ {
2116
+ "epoch": 0.536,
2117
+ "grad_norm": 0.1402324140071869,
2118
+ "learning_rate": 0.0004457827846023443,
2119
+ "loss": 3.3563668823242185,
2120
+ "step": 26800
2121
+ },
2122
+ {
2123
+ "epoch": 0.538,
2124
+ "grad_norm": 0.13993528485298157,
2125
+ "learning_rate": 0.0004426514160137837,
2126
+ "loss": 3.36125244140625,
2127
+ "step": 26900
2128
+ },
2129
+ {
2130
+ "epoch": 0.54,
2131
+ "grad_norm": 0.14961598813533783,
2132
+ "learning_rate": 0.000439522325094595,
2133
+ "loss": 3.360479431152344,
2134
+ "step": 27000
2135
+ },
2136
+ {
2137
+ "epoch": 0.54,
2138
+ "eval_accuracy": 0.37769785985999915,
2139
+ "eval_loss": 3.3393123149871826,
2140
+ "eval_runtime": 6.2827,
2141
+ "eval_samples_per_second": 308.945,
2142
+ "eval_steps_per_second": 19.418,
2143
+ "step": 27000
2144
+ },
2145
+ {
2146
+ "epoch": 0.542,
2147
+ "grad_norm": 0.12927019596099854,
2148
+ "learning_rate": 0.0004363956361204626,
2149
+ "loss": 3.3508065795898436,
2150
+ "step": 27100
2151
+ },
2152
+ {
2153
+ "epoch": 0.544,
2154
+ "grad_norm": 0.15256398916244507,
2155
+ "learning_rate": 0.0004332714732716751,
2156
+ "loss": 3.357308044433594,
2157
+ "step": 27200
2158
+ },
2159
+ {
2160
+ "epoch": 0.546,
2161
+ "grad_norm": 0.13779006898403168,
2162
+ "learning_rate": 0.0004301499606281931,
2163
+ "loss": 3.3775006103515626,
2164
+ "step": 27300
2165
+ },
2166
+ {
2167
+ "epoch": 0.548,
2168
+ "grad_norm": 0.1694018542766571,
2169
+ "learning_rate": 0.00042703122216472094,
2170
+ "loss": 3.3572964477539062,
2171
+ "step": 27400
2172
+ },
2173
+ {
2174
+ "epoch": 0.55,
2175
+ "grad_norm": 0.1522066593170166,
2176
+ "learning_rate": 0.00042391538174578263,
2177
+ "loss": 3.3765057373046874,
2178
+ "step": 27500
2179
+ },
2180
+ {
2181
+ "epoch": 0.552,
2182
+ "grad_norm": 0.14023618400096893,
2183
+ "learning_rate": 0.0004208025631208036,
2184
+ "loss": 3.355110778808594,
2185
+ "step": 27600
2186
+ },
2187
+ {
2188
+ "epoch": 0.554,
2189
+ "grad_norm": 0.14079532027244568,
2190
+ "learning_rate": 0.0004176928899191941,
2191
+ "loss": 3.3566073608398437,
2192
+ "step": 27700
2193
+ },
2194
+ {
2195
+ "epoch": 0.556,
2196
+ "grad_norm": 0.14286920428276062,
2197
+ "learning_rate": 0.0004145864856454404,
2198
+ "loss": 3.3700027465820312,
2199
+ "step": 27800
2200
+ },
2201
+ {
2202
+ "epoch": 0.558,
2203
+ "grad_norm": 0.13522571325302124,
2204
+ "learning_rate": 0.00041148347367419983,
2205
+ "loss": 3.3883221435546873,
2206
+ "step": 27900
2207
+ },
2208
+ {
2209
+ "epoch": 0.56,
2210
+ "grad_norm": 0.18117037415504456,
2211
+ "learning_rate": 0.00040838397724539947,
2212
+ "loss": 3.3707275390625,
2213
+ "step": 28000
2214
+ },
2215
+ {
2216
+ "epoch": 0.56,
2217
+ "eval_accuracy": 0.3787030511639349,
2218
+ "eval_loss": 3.330246925354004,
2219
+ "eval_runtime": 6.6261,
2220
+ "eval_samples_per_second": 292.932,
2221
+ "eval_steps_per_second": 18.412,
2222
+ "step": 28000
2223
  }
2224
  ],
2225
  "logging_steps": 100,
 
2239
  "attributes": {}
2240
  }
2241
  },
2242
+ "total_flos": 1.09758904270848e+18,
2243
  "train_batch_size": 120,
2244
  "trial_name": null,
2245
  "trial_params": null