Run 3. Outer Step 4. Inner Step 0. Peers 43.
Browse files- config.json +15 -15
- inner_optimizer.pt +1 -1
- model.safetensors +1 -1
- outer_optimizer.pt +1 -1
config.json
CHANGED
|
@@ -6,7 +6,7 @@
|
|
| 6 |
"1": "SUCCESS",
|
| 7 |
"10": "NON_PARTICIPATING",
|
| 8 |
"100": "NON_PARTICIPATING",
|
| 9 |
-
"101": "
|
| 10 |
"102": "SUCCESS",
|
| 11 |
"103": "NON_PARTICIPATING",
|
| 12 |
"104": "NON_PARTICIPATING",
|
|
@@ -14,7 +14,7 @@
|
|
| 14 |
"106": "NON_PARTICIPATING",
|
| 15 |
"107": "NON_PARTICIPATING",
|
| 16 |
"108": "NON_PARTICIPATING",
|
| 17 |
-
"109": "
|
| 18 |
"11": "NON_PARTICIPATING",
|
| 19 |
"110": "NON_PARTICIPATING",
|
| 20 |
"111": "NON_PARTICIPATING",
|
|
@@ -33,7 +33,7 @@
|
|
| 33 |
"123": "NON_PARTICIPATING",
|
| 34 |
"124": "NON_PARTICIPATING",
|
| 35 |
"125": "NON_PARTICIPATING",
|
| 36 |
-
"126": "
|
| 37 |
"127": "NON_PARTICIPATING",
|
| 38 |
"128": "NON_PARTICIPATING",
|
| 39 |
"129": "NON_PARTICIPATING",
|
|
@@ -75,9 +75,9 @@
|
|
| 75 |
"161": "NON_PARTICIPATING",
|
| 76 |
"162": "NON_PARTICIPATING",
|
| 77 |
"163": "NON_PARTICIPATING",
|
| 78 |
-
"164": "
|
| 79 |
-
"165": "
|
| 80 |
-
"166": "
|
| 81 |
"167": "NON_PARTICIPATING",
|
| 82 |
"168": "NON_PARTICIPATING",
|
| 83 |
"169": "NON_PARTICIPATING",
|
|
@@ -137,7 +137,7 @@
|
|
| 137 |
"217": "NON_PARTICIPATING",
|
| 138 |
"218": "SUCCESS",
|
| 139 |
"219": "NON_PARTICIPATING",
|
| 140 |
-
"22": "
|
| 141 |
"220": "NON_PARTICIPATING",
|
| 142 |
"221": "SUCCESS",
|
| 143 |
"222": "NON_PARTICIPATING",
|
|
@@ -148,13 +148,13 @@
|
|
| 148 |
"227": "NON_PARTICIPATING",
|
| 149 |
"228": "NON_PARTICIPATING",
|
| 150 |
"229": "NON_PARTICIPATING",
|
| 151 |
-
"23": "
|
| 152 |
"230": "NON_PARTICIPATING",
|
| 153 |
"231": "NON_PARTICIPATING",
|
| 154 |
"232": "NON_PARTICIPATING",
|
| 155 |
"233": "NON_PARTICIPATING",
|
| 156 |
"234": "NON_PARTICIPATING",
|
| 157 |
-
"235": "
|
| 158 |
"236": "SUCCESS",
|
| 159 |
"237": "NON_PARTICIPATING",
|
| 160 |
"238": "NON_PARTICIPATING",
|
|
@@ -187,7 +187,7 @@
|
|
| 187 |
"32": "SUCCESS",
|
| 188 |
"33": "NON_PARTICIPATING",
|
| 189 |
"34": "NON_PARTICIPATING",
|
| 190 |
-
"35": "
|
| 191 |
"36": "SUCCESS",
|
| 192 |
"37": "NON_PARTICIPATING",
|
| 193 |
"38": "NON_PARTICIPATING",
|
|
@@ -195,8 +195,8 @@
|
|
| 195 |
"4": "NON_PARTICIPATING",
|
| 196 |
"40": "NON_PARTICIPATING",
|
| 197 |
"41": "NON_PARTICIPATING",
|
| 198 |
-
"42": "
|
| 199 |
-
"43": "
|
| 200 |
"44": "NON_PARTICIPATING",
|
| 201 |
"45": "NON_PARTICIPATING",
|
| 202 |
"46": "NON_PARTICIPATING",
|
|
@@ -206,7 +206,7 @@
|
|
| 206 |
"5": "NON_PARTICIPATING",
|
| 207 |
"50": "SUCCESS",
|
| 208 |
"51": "NON_PARTICIPATING",
|
| 209 |
-
"52": "
|
| 210 |
"53": "NON_PARTICIPATING",
|
| 211 |
"54": "NON_PARTICIPATING",
|
| 212 |
"55": "NON_PARTICIPATING",
|
|
@@ -222,7 +222,7 @@
|
|
| 222 |
"64": "NON_PARTICIPATING",
|
| 223 |
"65": "NON_PARTICIPATING",
|
| 224 |
"66": "NON_PARTICIPATING",
|
| 225 |
-
"67": "
|
| 226 |
"68": "NON_PARTICIPATING",
|
| 227 |
"69": "NON_PARTICIPATING",
|
| 228 |
"7": "NON_PARTICIPATING",
|
|
@@ -275,7 +275,7 @@
|
|
| 275 |
"initializer_range": 0.02,
|
| 276 |
"inner_step": 0,
|
| 277 |
"inner_steps": 0,
|
| 278 |
-
"last_allreduce_block":
|
| 279 |
"layer_norm_epsilon": 1e-05,
|
| 280 |
"model_type": "gpt_optimized",
|
| 281 |
"n_embd": 1280,
|
|
|
|
| 6 |
"1": "SUCCESS",
|
| 7 |
"10": "NON_PARTICIPATING",
|
| 8 |
"100": "NON_PARTICIPATING",
|
| 9 |
+
"101": "NON_PARTICIPATING",
|
| 10 |
"102": "SUCCESS",
|
| 11 |
"103": "NON_PARTICIPATING",
|
| 12 |
"104": "NON_PARTICIPATING",
|
|
|
|
| 14 |
"106": "NON_PARTICIPATING",
|
| 15 |
"107": "NON_PARTICIPATING",
|
| 16 |
"108": "NON_PARTICIPATING",
|
| 17 |
+
"109": "SUCCESS",
|
| 18 |
"11": "NON_PARTICIPATING",
|
| 19 |
"110": "NON_PARTICIPATING",
|
| 20 |
"111": "NON_PARTICIPATING",
|
|
|
|
| 33 |
"123": "NON_PARTICIPATING",
|
| 34 |
"124": "NON_PARTICIPATING",
|
| 35 |
"125": "NON_PARTICIPATING",
|
| 36 |
+
"126": "SUCCESS",
|
| 37 |
"127": "NON_PARTICIPATING",
|
| 38 |
"128": "NON_PARTICIPATING",
|
| 39 |
"129": "NON_PARTICIPATING",
|
|
|
|
| 75 |
"161": "NON_PARTICIPATING",
|
| 76 |
"162": "NON_PARTICIPATING",
|
| 77 |
"163": "NON_PARTICIPATING",
|
| 78 |
+
"164": "SUCCESS",
|
| 79 |
+
"165": "SUCCESS",
|
| 80 |
+
"166": "NON_PARTICIPATING",
|
| 81 |
"167": "NON_PARTICIPATING",
|
| 82 |
"168": "NON_PARTICIPATING",
|
| 83 |
"169": "NON_PARTICIPATING",
|
|
|
|
| 137 |
"217": "NON_PARTICIPATING",
|
| 138 |
"218": "SUCCESS",
|
| 139 |
"219": "NON_PARTICIPATING",
|
| 140 |
+
"22": "SUCCESS",
|
| 141 |
"220": "NON_PARTICIPATING",
|
| 142 |
"221": "SUCCESS",
|
| 143 |
"222": "NON_PARTICIPATING",
|
|
|
|
| 148 |
"227": "NON_PARTICIPATING",
|
| 149 |
"228": "NON_PARTICIPATING",
|
| 150 |
"229": "NON_PARTICIPATING",
|
| 151 |
+
"23": "SUCCESS",
|
| 152 |
"230": "NON_PARTICIPATING",
|
| 153 |
"231": "NON_PARTICIPATING",
|
| 154 |
"232": "NON_PARTICIPATING",
|
| 155 |
"233": "NON_PARTICIPATING",
|
| 156 |
"234": "NON_PARTICIPATING",
|
| 157 |
+
"235": "SUCCESS",
|
| 158 |
"236": "SUCCESS",
|
| 159 |
"237": "NON_PARTICIPATING",
|
| 160 |
"238": "NON_PARTICIPATING",
|
|
|
|
| 187 |
"32": "SUCCESS",
|
| 188 |
"33": "NON_PARTICIPATING",
|
| 189 |
"34": "NON_PARTICIPATING",
|
| 190 |
+
"35": "SUCCESS",
|
| 191 |
"36": "SUCCESS",
|
| 192 |
"37": "NON_PARTICIPATING",
|
| 193 |
"38": "NON_PARTICIPATING",
|
|
|
|
| 195 |
"4": "NON_PARTICIPATING",
|
| 196 |
"40": "NON_PARTICIPATING",
|
| 197 |
"41": "NON_PARTICIPATING",
|
| 198 |
+
"42": "SUCCESS",
|
| 199 |
+
"43": "SUCCESS",
|
| 200 |
"44": "NON_PARTICIPATING",
|
| 201 |
"45": "NON_PARTICIPATING",
|
| 202 |
"46": "NON_PARTICIPATING",
|
|
|
|
| 206 |
"5": "NON_PARTICIPATING",
|
| 207 |
"50": "SUCCESS",
|
| 208 |
"51": "NON_PARTICIPATING",
|
| 209 |
+
"52": "SUCCESS",
|
| 210 |
"53": "NON_PARTICIPATING",
|
| 211 |
"54": "NON_PARTICIPATING",
|
| 212 |
"55": "NON_PARTICIPATING",
|
|
|
|
| 222 |
"64": "NON_PARTICIPATING",
|
| 223 |
"65": "NON_PARTICIPATING",
|
| 224 |
"66": "NON_PARTICIPATING",
|
| 225 |
+
"67": "SUCCESS",
|
| 226 |
"68": "NON_PARTICIPATING",
|
| 227 |
"69": "NON_PARTICIPATING",
|
| 228 |
"7": "NON_PARTICIPATING",
|
|
|
|
| 275 |
"initializer_range": 0.02,
|
| 276 |
"inner_step": 0,
|
| 277 |
"inner_steps": 0,
|
| 278 |
+
"last_allreduce_block": 5373127,
|
| 279 |
"layer_norm_epsilon": 1e-05,
|
| 280 |
"model_type": "gpt_optimized",
|
| 281 |
"n_embd": 1280,
|
inner_optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 8081782026
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6aa3e615459f64a829bda9dc3ec6b510dfa330c12ef13ebdec51f7de8a2d19d0
|
| 3 |
size 8081782026
|
model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4040701744
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:26e3befdbe767fb0bcf344ed4e574e11ff254db1b24854bcbc93b45da46cf8d8
|
| 3 |
size 4040701744
|
outer_optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4040805354
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ac6e2e1980133853d29d20383a1d5d06a48716349938929d1800f09d48c00707
|
| 3 |
size 4040805354
|