Auto upload zain 2026-08-19T16:33:38.712916 (part 2)
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +1 -0
- zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/optimizer.pt +3 -0
- zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/rng_state.pth +3 -0
- zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/scheduler.pt +3 -0
- zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/tokenizer.json +0 -0
- zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/tokenizer_config.json +13 -0
- zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/trainer_state.json +1360 -0
- zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/training_args.bin +3 -0
- zain/Activation/out/mlp-linear-3L_run/checkpoint-400/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-400/optimizer.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-400/rng_state.pth +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-400/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-400/trainer_state.json +95 -111
- zain/Activation/out/mlp-linear-3L_run/checkpoint-400/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-500/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-500/optimizer.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-500/rng_state.pth +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-500/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-500/trainer_state.json +115 -139
- zain/Activation/out/mlp-linear-3L_run/checkpoint-500/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-600/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-600/optimizer.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-600/rng_state.pth +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-600/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-600/trainer_state.json +140 -164
- zain/Activation/out/mlp-linear-3L_run/checkpoint-600/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-700/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-700/optimizer.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-700/rng_state.pth +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-700/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-700/trainer_state.json +160 -192
- zain/Activation/out/mlp-linear-3L_run/checkpoint-700/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-800/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-800/optimizer.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-800/rng_state.pth +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-800/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-800/trainer_state.json +185 -217
- zain/Activation/out/mlp-linear-3L_run/checkpoint-800/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-900/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-900/optimizer.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-900/rng_state.pth +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-900/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-900/trainer_state.json +205 -245
- zain/Activation/out/mlp-linear-3L_run/checkpoint-900/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-3L_run/training_log.jsonl +0 -0
- zain/Activation/wandb/debug-internal.log +35 -63
- zain/Activation/wandb/debug.log +22 -24
- zain/Activation/wandb/run-20260819_163014-smtfejrp/files/output.log +364 -0
- zain/Activation/wandb/run-20260819_163014-smtfejrp/files/requirements.txt +149 -0
- zain/Activation/wandb/run-20260819_163014-smtfejrp/files/wandb-metadata.json +118 -0
.gitattributes
CHANGED
|
@@ -106,3 +106,4 @@ zain/Activation/wandb/run-20260818_234902-aoa8wkuw/run-aoa8wkuw.wandb filter=lfs
|
|
| 106 |
zain/Activation/wandb/run-20260818_235450-ckmqoyjn/run-ckmqoyjn.wandb filter=lfs diff=lfs merge=lfs -text
|
| 107 |
zain/Activation/wandb/run-20260819_000034-awq28nwd/run-awq28nwd.wandb filter=lfs diff=lfs merge=lfs -text
|
| 108 |
zain/Activation/wandb/run-20260819_000621-6zbmve2d/run-6zbmve2d.wandb filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 106 |
zain/Activation/wandb/run-20260818_235450-ckmqoyjn/run-ckmqoyjn.wandb filter=lfs diff=lfs merge=lfs -text
|
| 107 |
zain/Activation/wandb/run-20260819_000034-awq28nwd/run-awq28nwd.wandb filter=lfs diff=lfs merge=lfs -text
|
| 108 |
zain/Activation/wandb/run-20260819_000621-6zbmve2d/run-6zbmve2d.wandb filter=lfs diff=lfs merge=lfs -text
|
| 109 |
+
zain/Activation/wandb/run-20260819_163014-smtfejrp/run-smtfejrp.wandb filter=lfs diff=lfs merge=lfs -text
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ce76c8d713a0e55e136bdef2c6d778e39cc2c72664541e823421cce9f2b5e5e1
|
| 3 |
+
size 4089360
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9213080fe2b45399b87036ca9ff9164533abe6b368e5c828136ee184486749d4
|
| 3 |
+
size 14244
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:84a9254a847978e7f5aa3da452546981aa184489b48bcdca370761da60268da8
|
| 3 |
+
size 1064
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/tokenizer_config.json
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"backend": "tokenizers",
|
| 4 |
+
"bos_token": "<|endoftext|>",
|
| 5 |
+
"eos_token": "<|endoftext|>",
|
| 6 |
+
"errors": "replace",
|
| 7 |
+
"is_local": false,
|
| 8 |
+
"local_files_only": false,
|
| 9 |
+
"model_max_length": 1024,
|
| 10 |
+
"pad_token": "<|endoftext|>",
|
| 11 |
+
"tokenizer_class": "GPT2Tokenizer",
|
| 12 |
+
"unk_token": "<|endoftext|>"
|
| 13 |
+
}
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/trainer_state.json
ADDED
|
@@ -0,0 +1,1360 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": null,
|
| 3 |
+
"best_metric": null,
|
| 4 |
+
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.057296932928884395,
|
| 6 |
+
"eval_steps": 200,
|
| 7 |
+
"global_step": 3400,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.0003370407819346141,
|
| 14 |
+
"grad_norm": 1.4140625,
|
| 15 |
+
"learning_rate": 5.7e-06,
|
| 16 |
+
"loss": 8.324227142333985,
|
| 17 |
+
"step": 20
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"epoch": 0.0006740815638692282,
|
| 21 |
+
"grad_norm": 1.5390625,
|
| 22 |
+
"learning_rate": 1.17e-05,
|
| 23 |
+
"loss": 8.318060302734375,
|
| 24 |
+
"step": 40
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"epoch": 0.0010111223458038423,
|
| 28 |
+
"grad_norm": 1.609375,
|
| 29 |
+
"learning_rate": 1.7699999999999997e-05,
|
| 30 |
+
"loss": 8.293100738525391,
|
| 31 |
+
"step": 60
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"epoch": 0.0013481631277384564,
|
| 35 |
+
"grad_norm": 1.7578125,
|
| 36 |
+
"learning_rate": 2.3699999999999997e-05,
|
| 37 |
+
"loss": 8.227334594726562,
|
| 38 |
+
"step": 80
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
"epoch": 0.0016852039096730705,
|
| 42 |
+
"grad_norm": 1.5234375,
|
| 43 |
+
"learning_rate": 2.97e-05,
|
| 44 |
+
"loss": 8.111893463134766,
|
| 45 |
+
"step": 100
|
| 46 |
+
},
|
| 47 |
+
{
|
| 48 |
+
"epoch": 0.0020222446916076846,
|
| 49 |
+
"grad_norm": 1.3046875,
|
| 50 |
+
"learning_rate": 3.5699999999999994e-05,
|
| 51 |
+
"loss": 7.976696014404297,
|
| 52 |
+
"step": 120
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"epoch": 0.0023592854735422987,
|
| 56 |
+
"grad_norm": 1.296875,
|
| 57 |
+
"learning_rate": 4.17e-05,
|
| 58 |
+
"loss": 7.851339721679688,
|
| 59 |
+
"step": 140
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"epoch": 0.002696326255476913,
|
| 63 |
+
"grad_norm": 1.3046875,
|
| 64 |
+
"learning_rate": 4.7699999999999994e-05,
|
| 65 |
+
"loss": 7.724923706054687,
|
| 66 |
+
"step": 160
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"epoch": 0.003033367037411527,
|
| 70 |
+
"grad_norm": 1.3125,
|
| 71 |
+
"learning_rate": 5.369999999999999e-05,
|
| 72 |
+
"loss": 7.58428726196289,
|
| 73 |
+
"step": 180
|
| 74 |
+
},
|
| 75 |
+
{
|
| 76 |
+
"epoch": 0.003370407819346141,
|
| 77 |
+
"grad_norm": 1.25,
|
| 78 |
+
"learning_rate": 5.97e-05,
|
| 79 |
+
"loss": 7.439914703369141,
|
| 80 |
+
"step": 200
|
| 81 |
+
},
|
| 82 |
+
{
|
| 83 |
+
"epoch": 0.003370407819346141,
|
| 84 |
+
"eval_loss": 7.356490135192871,
|
| 85 |
+
"eval_runtime": 7.5119,
|
| 86 |
+
"eval_samples_per_second": 1268.257,
|
| 87 |
+
"eval_steps_per_second": 0.932,
|
| 88 |
+
"step": 200
|
| 89 |
+
},
|
| 90 |
+
{
|
| 91 |
+
"epoch": 0.003707448601280755,
|
| 92 |
+
"grad_norm": 1.25,
|
| 93 |
+
"learning_rate": 6.57e-05,
|
| 94 |
+
"loss": 7.283377075195313,
|
| 95 |
+
"step": 220
|
| 96 |
+
},
|
| 97 |
+
{
|
| 98 |
+
"epoch": 0.004044489383215369,
|
| 99 |
+
"grad_norm": 1.25,
|
| 100 |
+
"learning_rate": 7.17e-05,
|
| 101 |
+
"loss": 7.127851104736328,
|
| 102 |
+
"step": 240
|
| 103 |
+
},
|
| 104 |
+
{
|
| 105 |
+
"epoch": 0.004381530165149983,
|
| 106 |
+
"grad_norm": 1.2109375,
|
| 107 |
+
"learning_rate": 7.769999999999999e-05,
|
| 108 |
+
"loss": 6.964313507080078,
|
| 109 |
+
"step": 260
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"epoch": 0.0047185709470845974,
|
| 113 |
+
"grad_norm": 1.171875,
|
| 114 |
+
"learning_rate": 8.37e-05,
|
| 115 |
+
"loss": 6.807338714599609,
|
| 116 |
+
"step": 280
|
| 117 |
+
},
|
| 118 |
+
{
|
| 119 |
+
"epoch": 0.005055611729019211,
|
| 120 |
+
"grad_norm": 1.1328125,
|
| 121 |
+
"learning_rate": 8.969999999999998e-05,
|
| 122 |
+
"loss": 6.667655181884766,
|
| 123 |
+
"step": 300
|
| 124 |
+
},
|
| 125 |
+
{
|
| 126 |
+
"epoch": 0.005392652510953826,
|
| 127 |
+
"grad_norm": 1.109375,
|
| 128 |
+
"learning_rate": 9.57e-05,
|
| 129 |
+
"loss": 6.523377227783203,
|
| 130 |
+
"step": 320
|
| 131 |
+
},
|
| 132 |
+
{
|
| 133 |
+
"epoch": 0.005729693292888439,
|
| 134 |
+
"grad_norm": 1.1171875,
|
| 135 |
+
"learning_rate": 0.00010169999999999999,
|
| 136 |
+
"loss": 6.383005142211914,
|
| 137 |
+
"step": 340
|
| 138 |
+
},
|
| 139 |
+
{
|
| 140 |
+
"epoch": 0.006066734074823054,
|
| 141 |
+
"grad_norm": 1.6875,
|
| 142 |
+
"learning_rate": 0.00010769999999999999,
|
| 143 |
+
"loss": 6.261091232299805,
|
| 144 |
+
"step": 360
|
| 145 |
+
},
|
| 146 |
+
{
|
| 147 |
+
"epoch": 0.0064037748567576675,
|
| 148 |
+
"grad_norm": 1.140625,
|
| 149 |
+
"learning_rate": 0.00011369999999999999,
|
| 150 |
+
"loss": 6.122833251953125,
|
| 151 |
+
"step": 380
|
| 152 |
+
},
|
| 153 |
+
{
|
| 154 |
+
"epoch": 0.006740815638692282,
|
| 155 |
+
"grad_norm": 1.3984375,
|
| 156 |
+
"learning_rate": 0.0001197,
|
| 157 |
+
"loss": 6.019657897949219,
|
| 158 |
+
"step": 400
|
| 159 |
+
},
|
| 160 |
+
{
|
| 161 |
+
"epoch": 0.006740815638692282,
|
| 162 |
+
"eval_loss": 5.966014385223389,
|
| 163 |
+
"eval_runtime": 7.516,
|
| 164 |
+
"eval_samples_per_second": 1267.556,
|
| 165 |
+
"eval_steps_per_second": 0.931,
|
| 166 |
+
"step": 400
|
| 167 |
+
},
|
| 168 |
+
{
|
| 169 |
+
"epoch": 0.007077856420626896,
|
| 170 |
+
"grad_norm": 0.98046875,
|
| 171 |
+
"learning_rate": 0.0001257,
|
| 172 |
+
"loss": 5.9373779296875,
|
| 173 |
+
"step": 420
|
| 174 |
+
},
|
| 175 |
+
{
|
| 176 |
+
"epoch": 0.00741489720256151,
|
| 177 |
+
"grad_norm": 1.6328125,
|
| 178 |
+
"learning_rate": 0.00013169999999999998,
|
| 179 |
+
"loss": 5.839211273193359,
|
| 180 |
+
"step": 440
|
| 181 |
+
},
|
| 182 |
+
{
|
| 183 |
+
"epoch": 0.007751937984496124,
|
| 184 |
+
"grad_norm": 0.9609375,
|
| 185 |
+
"learning_rate": 0.00013769999999999999,
|
| 186 |
+
"loss": 5.740922927856445,
|
| 187 |
+
"step": 460
|
| 188 |
+
},
|
| 189 |
+
{
|
| 190 |
+
"epoch": 0.008088978766430738,
|
| 191 |
+
"grad_norm": 0.90234375,
|
| 192 |
+
"learning_rate": 0.00014369999999999997,
|
| 193 |
+
"loss": 5.6399181365966795,
|
| 194 |
+
"step": 480
|
| 195 |
+
},
|
| 196 |
+
{
|
| 197 |
+
"epoch": 0.008426019548365353,
|
| 198 |
+
"grad_norm": 1.1796875,
|
| 199 |
+
"learning_rate": 0.00014969999999999998,
|
| 200 |
+
"loss": 5.560699081420898,
|
| 201 |
+
"step": 500
|
| 202 |
+
},
|
| 203 |
+
{
|
| 204 |
+
"epoch": 0.008763060330299966,
|
| 205 |
+
"grad_norm": 2.328125,
|
| 206 |
+
"learning_rate": 0.0001557,
|
| 207 |
+
"loss": 5.474863433837891,
|
| 208 |
+
"step": 520
|
| 209 |
+
},
|
| 210 |
+
{
|
| 211 |
+
"epoch": 0.00910010111223458,
|
| 212 |
+
"grad_norm": 1.125,
|
| 213 |
+
"learning_rate": 0.0001617,
|
| 214 |
+
"loss": 5.396588516235352,
|
| 215 |
+
"step": 540
|
| 216 |
+
},
|
| 217 |
+
{
|
| 218 |
+
"epoch": 0.009437141894169195,
|
| 219 |
+
"grad_norm": 1.6484375,
|
| 220 |
+
"learning_rate": 0.0001677,
|
| 221 |
+
"loss": 5.331023406982422,
|
| 222 |
+
"step": 560
|
| 223 |
+
},
|
| 224 |
+
{
|
| 225 |
+
"epoch": 0.00977418267610381,
|
| 226 |
+
"grad_norm": 1.0703125,
|
| 227 |
+
"learning_rate": 0.00017369999999999997,
|
| 228 |
+
"loss": 5.257175445556641,
|
| 229 |
+
"step": 580
|
| 230 |
+
},
|
| 231 |
+
{
|
| 232 |
+
"epoch": 0.010111223458038422,
|
| 233 |
+
"grad_norm": 2.359375,
|
| 234 |
+
"learning_rate": 0.00017969999999999998,
|
| 235 |
+
"loss": 5.152382659912109,
|
| 236 |
+
"step": 600
|
| 237 |
+
},
|
| 238 |
+
{
|
| 239 |
+
"epoch": 0.010111223458038422,
|
| 240 |
+
"eval_loss": 5.1175713539123535,
|
| 241 |
+
"eval_runtime": 7.457,
|
| 242 |
+
"eval_samples_per_second": 1277.585,
|
| 243 |
+
"eval_steps_per_second": 0.939,
|
| 244 |
+
"step": 600
|
| 245 |
+
},
|
| 246 |
+
{
|
| 247 |
+
"epoch": 0.010448264239973037,
|
| 248 |
+
"grad_norm": 1.375,
|
| 249 |
+
"learning_rate": 0.0001857,
|
| 250 |
+
"loss": 5.096985244750977,
|
| 251 |
+
"step": 620
|
| 252 |
+
},
|
| 253 |
+
{
|
| 254 |
+
"epoch": 0.010785305021907651,
|
| 255 |
+
"grad_norm": 1.421875,
|
| 256 |
+
"learning_rate": 0.0001917,
|
| 257 |
+
"loss": 5.019801330566406,
|
| 258 |
+
"step": 640
|
| 259 |
+
},
|
| 260 |
+
{
|
| 261 |
+
"epoch": 0.011122345803842264,
|
| 262 |
+
"grad_norm": 2.125,
|
| 263 |
+
"learning_rate": 0.00019769999999999998,
|
| 264 |
+
"loss": 4.979957962036133,
|
| 265 |
+
"step": 660
|
| 266 |
+
},
|
| 267 |
+
{
|
| 268 |
+
"epoch": 0.011459386585776879,
|
| 269 |
+
"grad_norm": 2.203125,
|
| 270 |
+
"learning_rate": 0.0002037,
|
| 271 |
+
"loss": 4.945652008056641,
|
| 272 |
+
"step": 680
|
| 273 |
+
},
|
| 274 |
+
{
|
| 275 |
+
"epoch": 0.011796427367711493,
|
| 276 |
+
"grad_norm": 1.8515625,
|
| 277 |
+
"learning_rate": 0.00020969999999999997,
|
| 278 |
+
"loss": 4.866051483154297,
|
| 279 |
+
"step": 700
|
| 280 |
+
},
|
| 281 |
+
{
|
| 282 |
+
"epoch": 0.012133468149646108,
|
| 283 |
+
"grad_norm": 1.625,
|
| 284 |
+
"learning_rate": 0.00021569999999999998,
|
| 285 |
+
"loss": 4.841766357421875,
|
| 286 |
+
"step": 720
|
| 287 |
+
},
|
| 288 |
+
{
|
| 289 |
+
"epoch": 0.01247050893158072,
|
| 290 |
+
"grad_norm": 3.40625,
|
| 291 |
+
"learning_rate": 0.00022169999999999997,
|
| 292 |
+
"loss": 4.798672103881836,
|
| 293 |
+
"step": 740
|
| 294 |
+
},
|
| 295 |
+
{
|
| 296 |
+
"epoch": 0.012807549713515335,
|
| 297 |
+
"grad_norm": 2.21875,
|
| 298 |
+
"learning_rate": 0.00022769999999999998,
|
| 299 |
+
"loss": 4.768531036376953,
|
| 300 |
+
"step": 760
|
| 301 |
+
},
|
| 302 |
+
{
|
| 303 |
+
"epoch": 0.01314459049544995,
|
| 304 |
+
"grad_norm": 1.640625,
|
| 305 |
+
"learning_rate": 0.0002337,
|
| 306 |
+
"loss": 4.73585319519043,
|
| 307 |
+
"step": 780
|
| 308 |
+
},
|
| 309 |
+
{
|
| 310 |
+
"epoch": 0.013481631277384564,
|
| 311 |
+
"grad_norm": 1.6875,
|
| 312 |
+
"learning_rate": 0.0002397,
|
| 313 |
+
"loss": 4.697021865844727,
|
| 314 |
+
"step": 800
|
| 315 |
+
},
|
| 316 |
+
{
|
| 317 |
+
"epoch": 0.013481631277384564,
|
| 318 |
+
"eval_loss": 4.676848411560059,
|
| 319 |
+
"eval_runtime": 7.4394,
|
| 320 |
+
"eval_samples_per_second": 1280.62,
|
| 321 |
+
"eval_steps_per_second": 0.941,
|
| 322 |
+
"step": 800
|
| 323 |
+
},
|
| 324 |
+
{
|
| 325 |
+
"epoch": 0.013818672059319177,
|
| 326 |
+
"grad_norm": 1.6484375,
|
| 327 |
+
"learning_rate": 0.00024569999999999995,
|
| 328 |
+
"loss": 4.658943176269531,
|
| 329 |
+
"step": 820
|
| 330 |
+
},
|
| 331 |
+
{
|
| 332 |
+
"epoch": 0.014155712841253791,
|
| 333 |
+
"grad_norm": 3.234375,
|
| 334 |
+
"learning_rate": 0.0002517,
|
| 335 |
+
"loss": 4.5957294464111325,
|
| 336 |
+
"step": 840
|
| 337 |
+
},
|
| 338 |
+
{
|
| 339 |
+
"epoch": 0.014492753623188406,
|
| 340 |
+
"grad_norm": 1.578125,
|
| 341 |
+
"learning_rate": 0.0002577,
|
| 342 |
+
"loss": 4.5967552185058596,
|
| 343 |
+
"step": 860
|
| 344 |
+
},
|
| 345 |
+
{
|
| 346 |
+
"epoch": 0.01482979440512302,
|
| 347 |
+
"grad_norm": 1.3203125,
|
| 348 |
+
"learning_rate": 0.00026369999999999996,
|
| 349 |
+
"loss": 4.549871444702148,
|
| 350 |
+
"step": 880
|
| 351 |
+
},
|
| 352 |
+
{
|
| 353 |
+
"epoch": 0.015166835187057633,
|
| 354 |
+
"grad_norm": 2.28125,
|
| 355 |
+
"learning_rate": 0.0002697,
|
| 356 |
+
"loss": 4.512709045410157,
|
| 357 |
+
"step": 900
|
| 358 |
+
},
|
| 359 |
+
{
|
| 360 |
+
"epoch": 0.015503875968992248,
|
| 361 |
+
"grad_norm": 1.6875,
|
| 362 |
+
"learning_rate": 0.0002757,
|
| 363 |
+
"loss": 4.475643920898437,
|
| 364 |
+
"step": 920
|
| 365 |
+
},
|
| 366 |
+
{
|
| 367 |
+
"epoch": 0.015840916750926862,
|
| 368 |
+
"grad_norm": 1.6015625,
|
| 369 |
+
"learning_rate": 0.00028169999999999996,
|
| 370 |
+
"loss": 4.43329086303711,
|
| 371 |
+
"step": 940
|
| 372 |
+
},
|
| 373 |
+
{
|
| 374 |
+
"epoch": 0.016177957532861477,
|
| 375 |
+
"grad_norm": 2.03125,
|
| 376 |
+
"learning_rate": 0.00028769999999999995,
|
| 377 |
+
"loss": 4.402384567260742,
|
| 378 |
+
"step": 960
|
| 379 |
+
},
|
| 380 |
+
{
|
| 381 |
+
"epoch": 0.01651499831479609,
|
| 382 |
+
"grad_norm": 1.71875,
|
| 383 |
+
"learning_rate": 0.0002937,
|
| 384 |
+
"loss": 4.382797622680664,
|
| 385 |
+
"step": 980
|
| 386 |
+
},
|
| 387 |
+
{
|
| 388 |
+
"epoch": 0.016852039096730706,
|
| 389 |
+
"grad_norm": 1.21875,
|
| 390 |
+
"learning_rate": 0.00029969999999999997,
|
| 391 |
+
"loss": 4.368251037597656,
|
| 392 |
+
"step": 1000
|
| 393 |
+
},
|
| 394 |
+
{
|
| 395 |
+
"epoch": 0.016852039096730706,
|
| 396 |
+
"eval_loss": 4.342709064483643,
|
| 397 |
+
"eval_runtime": 7.4513,
|
| 398 |
+
"eval_samples_per_second": 1278.567,
|
| 399 |
+
"eval_steps_per_second": 0.939,
|
| 400 |
+
"step": 1000
|
| 401 |
+
},
|
| 402 |
+
{
|
| 403 |
+
"epoch": 0.017189079878665317,
|
| 404 |
+
"grad_norm": 1.640625,
|
| 405 |
+
"learning_rate": 0.0003,
|
| 406 |
+
"loss": 4.33197135925293,
|
| 407 |
+
"step": 1020
|
| 408 |
+
},
|
| 409 |
+
{
|
| 410 |
+
"epoch": 0.01752612066059993,
|
| 411 |
+
"grad_norm": 1.8984375,
|
| 412 |
+
"learning_rate": 0.0003,
|
| 413 |
+
"loss": 4.315201950073242,
|
| 414 |
+
"step": 1040
|
| 415 |
+
},
|
| 416 |
+
{
|
| 417 |
+
"epoch": 0.017863161442534546,
|
| 418 |
+
"grad_norm": 1.8828125,
|
| 419 |
+
"learning_rate": 0.0003,
|
| 420 |
+
"loss": 4.262122344970703,
|
| 421 |
+
"step": 1060
|
| 422 |
+
},
|
| 423 |
+
{
|
| 424 |
+
"epoch": 0.01820020222446916,
|
| 425 |
+
"grad_norm": 2.3125,
|
| 426 |
+
"learning_rate": 0.0003,
|
| 427 |
+
"loss": 4.262465286254883,
|
| 428 |
+
"step": 1080
|
| 429 |
+
},
|
| 430 |
+
{
|
| 431 |
+
"epoch": 0.018537243006403775,
|
| 432 |
+
"grad_norm": 2.25,
|
| 433 |
+
"learning_rate": 0.0003,
|
| 434 |
+
"loss": 4.2220817565917965,
|
| 435 |
+
"step": 1100
|
| 436 |
+
},
|
| 437 |
+
{
|
| 438 |
+
"epoch": 0.01887428378833839,
|
| 439 |
+
"grad_norm": 1.6328125,
|
| 440 |
+
"learning_rate": 0.0003,
|
| 441 |
+
"loss": 4.21630973815918,
|
| 442 |
+
"step": 1120
|
| 443 |
+
},
|
| 444 |
+
{
|
| 445 |
+
"epoch": 0.019211324570273004,
|
| 446 |
+
"grad_norm": 1.84375,
|
| 447 |
+
"learning_rate": 0.0003,
|
| 448 |
+
"loss": 4.189186859130859,
|
| 449 |
+
"step": 1140
|
| 450 |
+
},
|
| 451 |
+
{
|
| 452 |
+
"epoch": 0.01954836535220762,
|
| 453 |
+
"grad_norm": 2.546875,
|
| 454 |
+
"learning_rate": 0.0003,
|
| 455 |
+
"loss": 4.164257431030274,
|
| 456 |
+
"step": 1160
|
| 457 |
+
},
|
| 458 |
+
{
|
| 459 |
+
"epoch": 0.01988540613414223,
|
| 460 |
+
"grad_norm": 1.4609375,
|
| 461 |
+
"learning_rate": 0.0003,
|
| 462 |
+
"loss": 4.161908721923828,
|
| 463 |
+
"step": 1180
|
| 464 |
+
},
|
| 465 |
+
{
|
| 466 |
+
"epoch": 0.020222446916076844,
|
| 467 |
+
"grad_norm": 1.5625,
|
| 468 |
+
"learning_rate": 0.0003,
|
| 469 |
+
"loss": 4.131203460693359,
|
| 470 |
+
"step": 1200
|
| 471 |
+
},
|
| 472 |
+
{
|
| 473 |
+
"epoch": 0.020222446916076844,
|
| 474 |
+
"eval_loss": 4.1340837478637695,
|
| 475 |
+
"eval_runtime": 7.4658,
|
| 476 |
+
"eval_samples_per_second": 1276.087,
|
| 477 |
+
"eval_steps_per_second": 0.938,
|
| 478 |
+
"step": 1200
|
| 479 |
+
},
|
| 480 |
+
{
|
| 481 |
+
"epoch": 0.02055948769801146,
|
| 482 |
+
"grad_norm": 1.2890625,
|
| 483 |
+
"learning_rate": 0.0003,
|
| 484 |
+
"loss": 4.109837341308594,
|
| 485 |
+
"step": 1220
|
| 486 |
+
},
|
| 487 |
+
{
|
| 488 |
+
"epoch": 0.020896528479946073,
|
| 489 |
+
"grad_norm": 2.109375,
|
| 490 |
+
"learning_rate": 0.0003,
|
| 491 |
+
"loss": 4.08643684387207,
|
| 492 |
+
"step": 1240
|
| 493 |
+
},
|
| 494 |
+
{
|
| 495 |
+
"epoch": 0.021233569261880688,
|
| 496 |
+
"grad_norm": 1.5859375,
|
| 497 |
+
"learning_rate": 0.0003,
|
| 498 |
+
"loss": 4.069121932983398,
|
| 499 |
+
"step": 1260
|
| 500 |
+
},
|
| 501 |
+
{
|
| 502 |
+
"epoch": 0.021570610043815303,
|
| 503 |
+
"grad_norm": 1.7890625,
|
| 504 |
+
"learning_rate": 0.0003,
|
| 505 |
+
"loss": 4.091477966308593,
|
| 506 |
+
"step": 1280
|
| 507 |
+
},
|
| 508 |
+
{
|
| 509 |
+
"epoch": 0.021907650825749917,
|
| 510 |
+
"grad_norm": 1.484375,
|
| 511 |
+
"learning_rate": 0.0003,
|
| 512 |
+
"loss": 4.061969757080078,
|
| 513 |
+
"step": 1300
|
| 514 |
+
},
|
| 515 |
+
{
|
| 516 |
+
"epoch": 0.022244691607684528,
|
| 517 |
+
"grad_norm": 1.5625,
|
| 518 |
+
"learning_rate": 0.0003,
|
| 519 |
+
"loss": 4.03832893371582,
|
| 520 |
+
"step": 1320
|
| 521 |
+
},
|
| 522 |
+
{
|
| 523 |
+
"epoch": 0.022581732389619143,
|
| 524 |
+
"grad_norm": 1.359375,
|
| 525 |
+
"learning_rate": 0.0003,
|
| 526 |
+
"loss": 4.031367492675781,
|
| 527 |
+
"step": 1340
|
| 528 |
+
},
|
| 529 |
+
{
|
| 530 |
+
"epoch": 0.022918773171553757,
|
| 531 |
+
"grad_norm": 1.4375,
|
| 532 |
+
"learning_rate": 0.0003,
|
| 533 |
+
"loss": 4.001505661010742,
|
| 534 |
+
"step": 1360
|
| 535 |
+
},
|
| 536 |
+
{
|
| 537 |
+
"epoch": 0.023255813953488372,
|
| 538 |
+
"grad_norm": 1.7734375,
|
| 539 |
+
"learning_rate": 0.0003,
|
| 540 |
+
"loss": 4.031853866577149,
|
| 541 |
+
"step": 1380
|
| 542 |
+
},
|
| 543 |
+
{
|
| 544 |
+
"epoch": 0.023592854735422986,
|
| 545 |
+
"grad_norm": 1.4609375,
|
| 546 |
+
"learning_rate": 0.0003,
|
| 547 |
+
"loss": 4.0036476135253904,
|
| 548 |
+
"step": 1400
|
| 549 |
+
},
|
| 550 |
+
{
|
| 551 |
+
"epoch": 0.023592854735422986,
|
| 552 |
+
"eval_loss": 3.999983072280884,
|
| 553 |
+
"eval_runtime": 7.5012,
|
| 554 |
+
"eval_samples_per_second": 1270.062,
|
| 555 |
+
"eval_steps_per_second": 0.933,
|
| 556 |
+
"step": 1400
|
| 557 |
+
},
|
| 558 |
+
{
|
| 559 |
+
"epoch": 0.0239298955173576,
|
| 560 |
+
"grad_norm": 1.53125,
|
| 561 |
+
"learning_rate": 0.0003,
|
| 562 |
+
"loss": 4.006050109863281,
|
| 563 |
+
"step": 1420
|
| 564 |
+
},
|
| 565 |
+
{
|
| 566 |
+
"epoch": 0.024266936299292215,
|
| 567 |
+
"grad_norm": 1.359375,
|
| 568 |
+
"learning_rate": 0.0003,
|
| 569 |
+
"loss": 3.9725921630859373,
|
| 570 |
+
"step": 1440
|
| 571 |
+
},
|
| 572 |
+
{
|
| 573 |
+
"epoch": 0.02460397708122683,
|
| 574 |
+
"grad_norm": 1.375,
|
| 575 |
+
"learning_rate": 0.0003,
|
| 576 |
+
"loss": 4.006849670410157,
|
| 577 |
+
"step": 1460
|
| 578 |
+
},
|
| 579 |
+
{
|
| 580 |
+
"epoch": 0.02494101786316144,
|
| 581 |
+
"grad_norm": 2.03125,
|
| 582 |
+
"learning_rate": 0.0003,
|
| 583 |
+
"loss": 3.955257797241211,
|
| 584 |
+
"step": 1480
|
| 585 |
+
},
|
| 586 |
+
{
|
| 587 |
+
"epoch": 0.025278058645096056,
|
| 588 |
+
"grad_norm": 1.359375,
|
| 589 |
+
"learning_rate": 0.0003,
|
| 590 |
+
"loss": 3.959492874145508,
|
| 591 |
+
"step": 1500
|
| 592 |
+
},
|
| 593 |
+
{
|
| 594 |
+
"epoch": 0.02561509942703067,
|
| 595 |
+
"grad_norm": 1.6484375,
|
| 596 |
+
"learning_rate": 0.0003,
|
| 597 |
+
"loss": 3.924856185913086,
|
| 598 |
+
"step": 1520
|
| 599 |
+
},
|
| 600 |
+
{
|
| 601 |
+
"epoch": 0.025952140208965285,
|
| 602 |
+
"grad_norm": 1.53125,
|
| 603 |
+
"learning_rate": 0.0003,
|
| 604 |
+
"loss": 3.915934753417969,
|
| 605 |
+
"step": 1540
|
| 606 |
+
},
|
| 607 |
+
{
|
| 608 |
+
"epoch": 0.0262891809908999,
|
| 609 |
+
"grad_norm": 1.46875,
|
| 610 |
+
"learning_rate": 0.0003,
|
| 611 |
+
"loss": 3.9002685546875,
|
| 612 |
+
"step": 1560
|
| 613 |
+
},
|
| 614 |
+
{
|
| 615 |
+
"epoch": 0.026626221772834514,
|
| 616 |
+
"grad_norm": 1.5078125,
|
| 617 |
+
"learning_rate": 0.0003,
|
| 618 |
+
"loss": 3.895922088623047,
|
| 619 |
+
"step": 1580
|
| 620 |
+
},
|
| 621 |
+
{
|
| 622 |
+
"epoch": 0.026963262554769128,
|
| 623 |
+
"grad_norm": 1.4453125,
|
| 624 |
+
"learning_rate": 0.0003,
|
| 625 |
+
"loss": 3.9393074035644533,
|
| 626 |
+
"step": 1600
|
| 627 |
+
},
|
| 628 |
+
{
|
| 629 |
+
"epoch": 0.026963262554769128,
|
| 630 |
+
"eval_loss": 3.9100229740142822,
|
| 631 |
+
"eval_runtime": 7.7821,
|
| 632 |
+
"eval_samples_per_second": 1224.215,
|
| 633 |
+
"eval_steps_per_second": 0.899,
|
| 634 |
+
"step": 1600
|
| 635 |
+
},
|
| 636 |
+
{
|
| 637 |
+
"epoch": 0.027300303336703743,
|
| 638 |
+
"grad_norm": 1.5234375,
|
| 639 |
+
"learning_rate": 0.0003,
|
| 640 |
+
"loss": 3.8935520172119142,
|
| 641 |
+
"step": 1620
|
| 642 |
+
},
|
| 643 |
+
{
|
| 644 |
+
"epoch": 0.027637344118638354,
|
| 645 |
+
"grad_norm": 1.65625,
|
| 646 |
+
"learning_rate": 0.0003,
|
| 647 |
+
"loss": 3.8741275787353517,
|
| 648 |
+
"step": 1640
|
| 649 |
+
},
|
| 650 |
+
{
|
| 651 |
+
"epoch": 0.02797438490057297,
|
| 652 |
+
"grad_norm": 1.8046875,
|
| 653 |
+
"learning_rate": 0.0003,
|
| 654 |
+
"loss": 3.8751964569091797,
|
| 655 |
+
"step": 1660
|
| 656 |
+
},
|
| 657 |
+
{
|
| 658 |
+
"epoch": 0.028311425682507583,
|
| 659 |
+
"grad_norm": 1.515625,
|
| 660 |
+
"learning_rate": 0.0003,
|
| 661 |
+
"loss": 3.8847869873046874,
|
| 662 |
+
"step": 1680
|
| 663 |
+
},
|
| 664 |
+
{
|
| 665 |
+
"epoch": 0.028648466464442197,
|
| 666 |
+
"grad_norm": 1.640625,
|
| 667 |
+
"learning_rate": 0.0003,
|
| 668 |
+
"loss": 3.900635528564453,
|
| 669 |
+
"step": 1700
|
| 670 |
+
},
|
| 671 |
+
{
|
| 672 |
+
"epoch": 0.028985507246376812,
|
| 673 |
+
"grad_norm": 1.359375,
|
| 674 |
+
"learning_rate": 0.0003,
|
| 675 |
+
"loss": 3.913836669921875,
|
| 676 |
+
"step": 1720
|
| 677 |
+
},
|
| 678 |
+
{
|
| 679 |
+
"epoch": 0.029322548028311426,
|
| 680 |
+
"grad_norm": 1.4921875,
|
| 681 |
+
"learning_rate": 0.0003,
|
| 682 |
+
"loss": 3.8652713775634764,
|
| 683 |
+
"step": 1740
|
| 684 |
+
},
|
| 685 |
+
{
|
| 686 |
+
"epoch": 0.02965958881024604,
|
| 687 |
+
"grad_norm": 1.921875,
|
| 688 |
+
"learning_rate": 0.0003,
|
| 689 |
+
"loss": 3.865024185180664,
|
| 690 |
+
"step": 1760
|
| 691 |
+
},
|
| 692 |
+
{
|
| 693 |
+
"epoch": 0.029996629592180656,
|
| 694 |
+
"grad_norm": 1.78125,
|
| 695 |
+
"learning_rate": 0.0003,
|
| 696 |
+
"loss": 3.825577163696289,
|
| 697 |
+
"step": 1780
|
| 698 |
+
},
|
| 699 |
+
{
|
| 700 |
+
"epoch": 0.030333670374115267,
|
| 701 |
+
"grad_norm": 1.3984375,
|
| 702 |
+
"learning_rate": 0.0003,
|
| 703 |
+
"loss": 3.842971420288086,
|
| 704 |
+
"step": 1800
|
| 705 |
+
},
|
| 706 |
+
{
|
| 707 |
+
"epoch": 0.030333670374115267,
|
| 708 |
+
"eval_loss": 3.8462002277374268,
|
| 709 |
+
"eval_runtime": 7.4378,
|
| 710 |
+
"eval_samples_per_second": 1280.896,
|
| 711 |
+
"eval_steps_per_second": 0.941,
|
| 712 |
+
"step": 1800
|
| 713 |
+
},
|
| 714 |
+
{
|
| 715 |
+
"epoch": 0.03067071115604988,
|
| 716 |
+
"grad_norm": 1.2890625,
|
| 717 |
+
"learning_rate": 0.0003,
|
| 718 |
+
"loss": 3.8543495178222655,
|
| 719 |
+
"step": 1820
|
| 720 |
+
},
|
| 721 |
+
{
|
| 722 |
+
"epoch": 0.031007751937984496,
|
| 723 |
+
"grad_norm": 1.53125,
|
| 724 |
+
"learning_rate": 0.0003,
|
| 725 |
+
"loss": 3.843314361572266,
|
| 726 |
+
"step": 1840
|
| 727 |
+
},
|
| 728 |
+
{
|
| 729 |
+
"epoch": 0.03134479271991911,
|
| 730 |
+
"grad_norm": 1.640625,
|
| 731 |
+
"learning_rate": 0.0003,
|
| 732 |
+
"loss": 3.841319274902344,
|
| 733 |
+
"step": 1860
|
| 734 |
+
},
|
| 735 |
+
{
|
| 736 |
+
"epoch": 0.031681833501853725,
|
| 737 |
+
"grad_norm": 1.109375,
|
| 738 |
+
"learning_rate": 0.0003,
|
| 739 |
+
"loss": 3.831389617919922,
|
| 740 |
+
"step": 1880
|
| 741 |
+
},
|
| 742 |
+
{
|
| 743 |
+
"epoch": 0.032018874283788336,
|
| 744 |
+
"grad_norm": 1.4140625,
|
| 745 |
+
"learning_rate": 0.0003,
|
| 746 |
+
"loss": 3.8035839080810545,
|
| 747 |
+
"step": 1900
|
| 748 |
+
},
|
| 749 |
+
{
|
| 750 |
+
"epoch": 0.032355915065722954,
|
| 751 |
+
"grad_norm": 1.8359375,
|
| 752 |
+
"learning_rate": 0.0003,
|
| 753 |
+
"loss": 3.8156478881835936,
|
| 754 |
+
"step": 1920
|
| 755 |
+
},
|
| 756 |
+
{
|
| 757 |
+
"epoch": 0.032692955847657565,
|
| 758 |
+
"grad_norm": 1.5,
|
| 759 |
+
"learning_rate": 0.0003,
|
| 760 |
+
"loss": 3.7988842010498045,
|
| 761 |
+
"step": 1940
|
| 762 |
+
},
|
| 763 |
+
{
|
| 764 |
+
"epoch": 0.03302999662959218,
|
| 765 |
+
"grad_norm": 2.078125,
|
| 766 |
+
"learning_rate": 0.0003,
|
| 767 |
+
"loss": 3.8014480590820314,
|
| 768 |
+
"step": 1960
|
| 769 |
+
},
|
| 770 |
+
{
|
| 771 |
+
"epoch": 0.033367037411526794,
|
| 772 |
+
"grad_norm": 1.640625,
|
| 773 |
+
"learning_rate": 0.0003,
|
| 774 |
+
"loss": 3.7902145385742188,
|
| 775 |
+
"step": 1980
|
| 776 |
+
},
|
| 777 |
+
{
|
| 778 |
+
"epoch": 0.03370407819346141,
|
| 779 |
+
"grad_norm": 1.6953125,
|
| 780 |
+
"learning_rate": 0.0003,
|
| 781 |
+
"loss": 3.8097972869873047,
|
| 782 |
+
"step": 2000
|
| 783 |
+
},
|
| 784 |
+
{
|
| 785 |
+
"epoch": 0.03370407819346141,
|
| 786 |
+
"eval_loss": 3.7975757122039795,
|
| 787 |
+
"eval_runtime": 7.4856,
|
| 788 |
+
"eval_samples_per_second": 1272.718,
|
| 789 |
+
"eval_steps_per_second": 0.935,
|
| 790 |
+
"step": 2000
|
| 791 |
+
},
|
| 792 |
+
{
|
| 793 |
+
"epoch": 0.03404111897539602,
|
| 794 |
+
"grad_norm": 1.7734375,
|
| 795 |
+
"learning_rate": 0.0003,
|
| 796 |
+
"loss": 3.801420211791992,
|
| 797 |
+
"step": 2020
|
| 798 |
+
},
|
| 799 |
+
{
|
| 800 |
+
"epoch": 0.034378159757330634,
|
| 801 |
+
"grad_norm": 1.5234375,
|
| 802 |
+
"learning_rate": 0.0003,
|
| 803 |
+
"loss": 3.787317657470703,
|
| 804 |
+
"step": 2040
|
| 805 |
+
},
|
| 806 |
+
{
|
| 807 |
+
"epoch": 0.03471520053926525,
|
| 808 |
+
"grad_norm": 1.46875,
|
| 809 |
+
"learning_rate": 0.0003,
|
| 810 |
+
"loss": 3.797407531738281,
|
| 811 |
+
"step": 2060
|
| 812 |
+
},
|
| 813 |
+
{
|
| 814 |
+
"epoch": 0.03505224132119986,
|
| 815 |
+
"grad_norm": 1.734375,
|
| 816 |
+
"learning_rate": 0.0003,
|
| 817 |
+
"loss": 3.759593963623047,
|
| 818 |
+
"step": 2080
|
| 819 |
+
},
|
| 820 |
+
{
|
| 821 |
+
"epoch": 0.03538928210313448,
|
| 822 |
+
"grad_norm": 1.4765625,
|
| 823 |
+
"learning_rate": 0.0003,
|
| 824 |
+
"loss": 3.7659400939941405,
|
| 825 |
+
"step": 2100
|
| 826 |
+
},
|
| 827 |
+
{
|
| 828 |
+
"epoch": 0.03572632288506909,
|
| 829 |
+
"grad_norm": 1.5,
|
| 830 |
+
"learning_rate": 0.0003,
|
| 831 |
+
"loss": 3.766071319580078,
|
| 832 |
+
"step": 2120
|
| 833 |
+
},
|
| 834 |
+
{
|
| 835 |
+
"epoch": 0.03606336366700371,
|
| 836 |
+
"grad_norm": 1.5078125,
|
| 837 |
+
"learning_rate": 0.0003,
|
| 838 |
+
"loss": 3.770547866821289,
|
| 839 |
+
"step": 2140
|
| 840 |
+
},
|
| 841 |
+
{
|
| 842 |
+
"epoch": 0.03640040444893832,
|
| 843 |
+
"grad_norm": 1.9453125,
|
| 844 |
+
"learning_rate": 0.0003,
|
| 845 |
+
"loss": 3.7436546325683593,
|
| 846 |
+
"step": 2160
|
| 847 |
+
},
|
| 848 |
+
{
|
| 849 |
+
"epoch": 0.03673744523087293,
|
| 850 |
+
"grad_norm": 1.390625,
|
| 851 |
+
"learning_rate": 0.0003,
|
| 852 |
+
"loss": 3.752705764770508,
|
| 853 |
+
"step": 2180
|
| 854 |
+
},
|
| 855 |
+
{
|
| 856 |
+
"epoch": 0.03707448601280755,
|
| 857 |
+
"grad_norm": 1.515625,
|
| 858 |
+
"learning_rate": 0.0003,
|
| 859 |
+
"loss": 3.7526622772216798,
|
| 860 |
+
"step": 2200
|
| 861 |
+
},
|
| 862 |
+
{
|
| 863 |
+
"epoch": 0.03707448601280755,
|
| 864 |
+
"eval_loss": 3.7623653411865234,
|
| 865 |
+
"eval_runtime": 7.4294,
|
| 866 |
+
"eval_samples_per_second": 1282.342,
|
| 867 |
+
"eval_steps_per_second": 0.942,
|
| 868 |
+
"step": 2200
|
| 869 |
+
},
|
| 870 |
+
{
|
| 871 |
+
"epoch": 0.03741152679474216,
|
| 872 |
+
"grad_norm": 1.765625,
|
| 873 |
+
"learning_rate": 0.0003,
|
| 874 |
+
"loss": 3.7532962799072265,
|
| 875 |
+
"step": 2220
|
| 876 |
+
},
|
| 877 |
+
{
|
| 878 |
+
"epoch": 0.03774856757667678,
|
| 879 |
+
"grad_norm": 1.765625,
|
| 880 |
+
"learning_rate": 0.0003,
|
| 881 |
+
"loss": 3.7590545654296874,
|
| 882 |
+
"step": 2240
|
| 883 |
+
},
|
| 884 |
+
{
|
| 885 |
+
"epoch": 0.03808560835861139,
|
| 886 |
+
"grad_norm": 1.71875,
|
| 887 |
+
"learning_rate": 0.0003,
|
| 888 |
+
"loss": 3.7438419342041014,
|
| 889 |
+
"step": 2260
|
| 890 |
+
},
|
| 891 |
+
{
|
| 892 |
+
"epoch": 0.03842264914054601,
|
| 893 |
+
"grad_norm": 1.859375,
|
| 894 |
+
"learning_rate": 0.0003,
|
| 895 |
+
"loss": 3.7286632537841795,
|
| 896 |
+
"step": 2280
|
| 897 |
+
},
|
| 898 |
+
{
|
| 899 |
+
"epoch": 0.03875968992248062,
|
| 900 |
+
"grad_norm": 1.609375,
|
| 901 |
+
"learning_rate": 0.0003,
|
| 902 |
+
"loss": 3.749654006958008,
|
| 903 |
+
"step": 2300
|
| 904 |
+
},
|
| 905 |
+
{
|
| 906 |
+
"epoch": 0.03909673070441524,
|
| 907 |
+
"grad_norm": 1.515625,
|
| 908 |
+
"learning_rate": 0.0003,
|
| 909 |
+
"loss": 3.716779327392578,
|
| 910 |
+
"step": 2320
|
| 911 |
+
},
|
| 912 |
+
{
|
| 913 |
+
"epoch": 0.03943377148634985,
|
| 914 |
+
"grad_norm": 1.34375,
|
| 915 |
+
"learning_rate": 0.0003,
|
| 916 |
+
"loss": 3.7487411499023438,
|
| 917 |
+
"step": 2340
|
| 918 |
+
},
|
| 919 |
+
{
|
| 920 |
+
"epoch": 0.03977081226828446,
|
| 921 |
+
"grad_norm": 1.5859375,
|
| 922 |
+
"learning_rate": 0.0003,
|
| 923 |
+
"loss": 3.748704528808594,
|
| 924 |
+
"step": 2360
|
| 925 |
+
},
|
| 926 |
+
{
|
| 927 |
+
"epoch": 0.04010785305021908,
|
| 928 |
+
"grad_norm": 1.421875,
|
| 929 |
+
"learning_rate": 0.0003,
|
| 930 |
+
"loss": 3.721243667602539,
|
| 931 |
+
"step": 2380
|
| 932 |
+
},
|
| 933 |
+
{
|
| 934 |
+
"epoch": 0.04044489383215369,
|
| 935 |
+
"grad_norm": 1.7109375,
|
| 936 |
+
"learning_rate": 0.0003,
|
| 937 |
+
"loss": 3.723830795288086,
|
| 938 |
+
"step": 2400
|
| 939 |
+
},
|
| 940 |
+
{
|
| 941 |
+
"epoch": 0.04044489383215369,
|
| 942 |
+
"eval_loss": 3.7325456142425537,
|
| 943 |
+
"eval_runtime": 7.4184,
|
| 944 |
+
"eval_samples_per_second": 1284.24,
|
| 945 |
+
"eval_steps_per_second": 0.944,
|
| 946 |
+
"step": 2400
|
| 947 |
+
},
|
| 948 |
+
{
|
| 949 |
+
"epoch": 0.04078193461408831,
|
| 950 |
+
"grad_norm": 1.546875,
|
| 951 |
+
"learning_rate": 0.0003,
|
| 952 |
+
"loss": 3.7124366760253906,
|
| 953 |
+
"step": 2420
|
| 954 |
+
},
|
| 955 |
+
{
|
| 956 |
+
"epoch": 0.04111897539602292,
|
| 957 |
+
"grad_norm": 1.546875,
|
| 958 |
+
"learning_rate": 0.0003,
|
| 959 |
+
"loss": 3.695796585083008,
|
| 960 |
+
"step": 2440
|
| 961 |
+
},
|
| 962 |
+
{
|
| 963 |
+
"epoch": 0.041456016177957536,
|
| 964 |
+
"grad_norm": 1.5078125,
|
| 965 |
+
"learning_rate": 0.0003,
|
| 966 |
+
"loss": 3.700525665283203,
|
| 967 |
+
"step": 2460
|
| 968 |
+
},
|
| 969 |
+
{
|
| 970 |
+
"epoch": 0.04179305695989215,
|
| 971 |
+
"grad_norm": 1.328125,
|
| 972 |
+
"learning_rate": 0.0003,
|
| 973 |
+
"loss": 3.7314002990722654,
|
| 974 |
+
"step": 2480
|
| 975 |
+
},
|
| 976 |
+
{
|
| 977 |
+
"epoch": 0.04213009774182676,
|
| 978 |
+
"grad_norm": 1.7421875,
|
| 979 |
+
"learning_rate": 0.0003,
|
| 980 |
+
"loss": 3.728242874145508,
|
| 981 |
+
"step": 2500
|
| 982 |
+
},
|
| 983 |
+
{
|
| 984 |
+
"epoch": 0.042467138523761376,
|
| 985 |
+
"grad_norm": 1.3125,
|
| 986 |
+
"learning_rate": 0.0003,
|
| 987 |
+
"loss": 3.7244899749755858,
|
| 988 |
+
"step": 2520
|
| 989 |
+
},
|
| 990 |
+
{
|
| 991 |
+
"epoch": 0.04280417930569599,
|
| 992 |
+
"grad_norm": 1.7265625,
|
| 993 |
+
"learning_rate": 0.0003,
|
| 994 |
+
"loss": 3.7204193115234374,
|
| 995 |
+
"step": 2540
|
| 996 |
+
},
|
| 997 |
+
{
|
| 998 |
+
"epoch": 0.043141220087630605,
|
| 999 |
+
"grad_norm": 2.09375,
|
| 1000 |
+
"learning_rate": 0.0003,
|
| 1001 |
+
"loss": 3.7023361206054686,
|
| 1002 |
+
"step": 2560
|
| 1003 |
+
},
|
| 1004 |
+
{
|
| 1005 |
+
"epoch": 0.043478260869565216,
|
| 1006 |
+
"grad_norm": 1.7890625,
|
| 1007 |
+
"learning_rate": 0.0003,
|
| 1008 |
+
"loss": 3.712936019897461,
|
| 1009 |
+
"step": 2580
|
| 1010 |
+
},
|
| 1011 |
+
{
|
| 1012 |
+
"epoch": 0.043815301651499834,
|
| 1013 |
+
"grad_norm": 1.640625,
|
| 1014 |
+
"learning_rate": 0.0003,
|
| 1015 |
+
"loss": 3.6954383850097656,
|
| 1016 |
+
"step": 2600
|
| 1017 |
+
},
|
| 1018 |
+
{
|
| 1019 |
+
"epoch": 0.043815301651499834,
|
| 1020 |
+
"eval_loss": 3.698262929916382,
|
| 1021 |
+
"eval_runtime": 7.6077,
|
| 1022 |
+
"eval_samples_per_second": 1252.285,
|
| 1023 |
+
"eval_steps_per_second": 0.92,
|
| 1024 |
+
"step": 2600
|
| 1025 |
+
},
|
| 1026 |
+
{
|
| 1027 |
+
"epoch": 0.044152342433434445,
|
| 1028 |
+
"grad_norm": 1.2890625,
|
| 1029 |
+
"learning_rate": 0.0003,
|
| 1030 |
+
"loss": 3.6608612060546877,
|
| 1031 |
+
"step": 2620
|
| 1032 |
+
},
|
| 1033 |
+
{
|
| 1034 |
+
"epoch": 0.044489383215369056,
|
| 1035 |
+
"grad_norm": 1.703125,
|
| 1036 |
+
"learning_rate": 0.0003,
|
| 1037 |
+
"loss": 3.68970947265625,
|
| 1038 |
+
"step": 2640
|
| 1039 |
+
},
|
| 1040 |
+
{
|
| 1041 |
+
"epoch": 0.044826423997303674,
|
| 1042 |
+
"grad_norm": 1.5234375,
|
| 1043 |
+
"learning_rate": 0.0003,
|
| 1044 |
+
"loss": 3.728267288208008,
|
| 1045 |
+
"step": 2660
|
| 1046 |
+
},
|
| 1047 |
+
{
|
| 1048 |
+
"epoch": 0.045163464779238285,
|
| 1049 |
+
"grad_norm": 1.59375,
|
| 1050 |
+
"learning_rate": 0.0003,
|
| 1051 |
+
"loss": 3.6732406616210938,
|
| 1052 |
+
"step": 2680
|
| 1053 |
+
},
|
| 1054 |
+
{
|
| 1055 |
+
"epoch": 0.0455005055611729,
|
| 1056 |
+
"grad_norm": 1.515625,
|
| 1057 |
+
"learning_rate": 0.0003,
|
| 1058 |
+
"loss": 3.679242706298828,
|
| 1059 |
+
"step": 2700
|
| 1060 |
+
},
|
| 1061 |
+
{
|
| 1062 |
+
"epoch": 0.045837546343107514,
|
| 1063 |
+
"grad_norm": 1.4921875,
|
| 1064 |
+
"learning_rate": 0.0003,
|
| 1065 |
+
"loss": 3.689365768432617,
|
| 1066 |
+
"step": 2720
|
| 1067 |
+
},
|
| 1068 |
+
{
|
| 1069 |
+
"epoch": 0.04617458712504213,
|
| 1070 |
+
"grad_norm": 1.3828125,
|
| 1071 |
+
"learning_rate": 0.0003,
|
| 1072 |
+
"loss": 3.6569408416748046,
|
| 1073 |
+
"step": 2740
|
| 1074 |
+
},
|
| 1075 |
+
{
|
| 1076 |
+
"epoch": 0.046511627906976744,
|
| 1077 |
+
"grad_norm": 1.8125,
|
| 1078 |
+
"learning_rate": 0.0003,
|
| 1079 |
+
"loss": 3.707513427734375,
|
| 1080 |
+
"step": 2760
|
| 1081 |
+
},
|
| 1082 |
+
{
|
| 1083 |
+
"epoch": 0.04684866868891136,
|
| 1084 |
+
"grad_norm": 1.828125,
|
| 1085 |
+
"learning_rate": 0.0003,
|
| 1086 |
+
"loss": 3.669821929931641,
|
| 1087 |
+
"step": 2780
|
| 1088 |
+
},
|
| 1089 |
+
{
|
| 1090 |
+
"epoch": 0.04718570947084597,
|
| 1091 |
+
"grad_norm": 1.4765625,
|
| 1092 |
+
"learning_rate": 0.0003,
|
| 1093 |
+
"loss": 3.652143859863281,
|
| 1094 |
+
"step": 2800
|
| 1095 |
+
},
|
| 1096 |
+
{
|
| 1097 |
+
"epoch": 0.04718570947084597,
|
| 1098 |
+
"eval_loss": 3.671437978744507,
|
| 1099 |
+
"eval_runtime": 7.4168,
|
| 1100 |
+
"eval_samples_per_second": 1284.522,
|
| 1101 |
+
"eval_steps_per_second": 0.944,
|
| 1102 |
+
"step": 2800
|
| 1103 |
+
},
|
| 1104 |
+
{
|
| 1105 |
+
"epoch": 0.047522750252780584,
|
| 1106 |
+
"grad_norm": 1.171875,
|
| 1107 |
+
"learning_rate": 0.0003,
|
| 1108 |
+
"loss": 3.684069061279297,
|
| 1109 |
+
"step": 2820
|
| 1110 |
+
},
|
| 1111 |
+
{
|
| 1112 |
+
"epoch": 0.0478597910347152,
|
| 1113 |
+
"grad_norm": 1.671875,
|
| 1114 |
+
"learning_rate": 0.0003,
|
| 1115 |
+
"loss": 3.6512351989746095,
|
| 1116 |
+
"step": 2840
|
| 1117 |
+
},
|
| 1118 |
+
{
|
| 1119 |
+
"epoch": 0.04819683181664981,
|
| 1120 |
+
"grad_norm": 1.8984375,
|
| 1121 |
+
"learning_rate": 0.0003,
|
| 1122 |
+
"loss": 3.6539031982421877,
|
| 1123 |
+
"step": 2860
|
| 1124 |
+
},
|
| 1125 |
+
{
|
| 1126 |
+
"epoch": 0.04853387259858443,
|
| 1127 |
+
"grad_norm": 1.46875,
|
| 1128 |
+
"learning_rate": 0.0003,
|
| 1129 |
+
"loss": 3.6580196380615235,
|
| 1130 |
+
"step": 2880
|
| 1131 |
+
},
|
| 1132 |
+
{
|
| 1133 |
+
"epoch": 0.04887091338051904,
|
| 1134 |
+
"grad_norm": 1.375,
|
| 1135 |
+
"learning_rate": 0.0003,
|
| 1136 |
+
"loss": 3.643080139160156,
|
| 1137 |
+
"step": 2900
|
| 1138 |
+
},
|
| 1139 |
+
{
|
| 1140 |
+
"epoch": 0.04920795416245366,
|
| 1141 |
+
"grad_norm": 1.8125,
|
| 1142 |
+
"learning_rate": 0.0003,
|
| 1143 |
+
"loss": 3.6346275329589846,
|
| 1144 |
+
"step": 2920
|
| 1145 |
+
},
|
| 1146 |
+
{
|
| 1147 |
+
"epoch": 0.04954499494438827,
|
| 1148 |
+
"grad_norm": 1.4921875,
|
| 1149 |
+
"learning_rate": 0.0003,
|
| 1150 |
+
"loss": 3.6610347747802736,
|
| 1151 |
+
"step": 2940
|
| 1152 |
+
},
|
| 1153 |
+
{
|
| 1154 |
+
"epoch": 0.04988203572632288,
|
| 1155 |
+
"grad_norm": 1.4453125,
|
| 1156 |
+
"learning_rate": 0.0003,
|
| 1157 |
+
"loss": 3.6628185272216798,
|
| 1158 |
+
"step": 2960
|
| 1159 |
+
},
|
| 1160 |
+
{
|
| 1161 |
+
"epoch": 0.0502190765082575,
|
| 1162 |
+
"grad_norm": 1.546875,
|
| 1163 |
+
"learning_rate": 0.0003,
|
| 1164 |
+
"loss": 3.6714599609375,
|
| 1165 |
+
"step": 2980
|
| 1166 |
+
},
|
| 1167 |
+
{
|
| 1168 |
+
"epoch": 0.05055611729019211,
|
| 1169 |
+
"grad_norm": 1.6796875,
|
| 1170 |
+
"learning_rate": 0.0003,
|
| 1171 |
+
"loss": 3.6391124725341797,
|
| 1172 |
+
"step": 3000
|
| 1173 |
+
},
|
| 1174 |
+
{
|
| 1175 |
+
"epoch": 0.05055611729019211,
|
| 1176 |
+
"eval_loss": 3.6535143852233887,
|
| 1177 |
+
"eval_runtime": 7.4227,
|
| 1178 |
+
"eval_samples_per_second": 1283.487,
|
| 1179 |
+
"eval_steps_per_second": 0.943,
|
| 1180 |
+
"step": 3000
|
| 1181 |
+
},
|
| 1182 |
+
{
|
| 1183 |
+
"epoch": 0.05089315807212673,
|
| 1184 |
+
"grad_norm": 1.453125,
|
| 1185 |
+
"learning_rate": 0.0003,
|
| 1186 |
+
"loss": 3.659154510498047,
|
| 1187 |
+
"step": 3020
|
| 1188 |
+
},
|
| 1189 |
+
{
|
| 1190 |
+
"epoch": 0.05123019885406134,
|
| 1191 |
+
"grad_norm": 1.6484375,
|
| 1192 |
+
"learning_rate": 0.0003,
|
| 1193 |
+
"loss": 3.6401947021484373,
|
| 1194 |
+
"step": 3040
|
| 1195 |
+
},
|
| 1196 |
+
{
|
| 1197 |
+
"epoch": 0.05156723963599596,
|
| 1198 |
+
"grad_norm": 1.4765625,
|
| 1199 |
+
"learning_rate": 0.0003,
|
| 1200 |
+
"loss": 3.67232666015625,
|
| 1201 |
+
"step": 3060
|
| 1202 |
+
},
|
| 1203 |
+
{
|
| 1204 |
+
"epoch": 0.05190428041793057,
|
| 1205 |
+
"grad_norm": 1.4296875,
|
| 1206 |
+
"learning_rate": 0.0003,
|
| 1207 |
+
"loss": 3.649647521972656,
|
| 1208 |
+
"step": 3080
|
| 1209 |
+
},
|
| 1210 |
+
{
|
| 1211 |
+
"epoch": 0.05224132119986518,
|
| 1212 |
+
"grad_norm": 1.796875,
|
| 1213 |
+
"learning_rate": 0.0003,
|
| 1214 |
+
"loss": 3.6890209197998045,
|
| 1215 |
+
"step": 3100
|
| 1216 |
+
},
|
| 1217 |
+
{
|
| 1218 |
+
"epoch": 0.0525783619817998,
|
| 1219 |
+
"grad_norm": 1.7109375,
|
| 1220 |
+
"learning_rate": 0.0003,
|
| 1221 |
+
"loss": 3.639459991455078,
|
| 1222 |
+
"step": 3120
|
| 1223 |
+
},
|
| 1224 |
+
{
|
| 1225 |
+
"epoch": 0.05291540276373441,
|
| 1226 |
+
"grad_norm": 1.828125,
|
| 1227 |
+
"learning_rate": 0.0003,
|
| 1228 |
+
"loss": 3.6333686828613283,
|
| 1229 |
+
"step": 3140
|
| 1230 |
+
},
|
| 1231 |
+
{
|
| 1232 |
+
"epoch": 0.05325244354566903,
|
| 1233 |
+
"grad_norm": 1.4609375,
|
| 1234 |
+
"learning_rate": 0.0003,
|
| 1235 |
+
"loss": 3.6446548461914063,
|
| 1236 |
+
"step": 3160
|
| 1237 |
+
},
|
| 1238 |
+
{
|
| 1239 |
+
"epoch": 0.05358948432760364,
|
| 1240 |
+
"grad_norm": 1.75,
|
| 1241 |
+
"learning_rate": 0.0003,
|
| 1242 |
+
"loss": 3.639226531982422,
|
| 1243 |
+
"step": 3180
|
| 1244 |
+
},
|
| 1245 |
+
{
|
| 1246 |
+
"epoch": 0.053926525109538256,
|
| 1247 |
+
"grad_norm": 1.3984375,
|
| 1248 |
+
"learning_rate": 0.0003,
|
| 1249 |
+
"loss": 3.6387962341308593,
|
| 1250 |
+
"step": 3200
|
| 1251 |
+
},
|
| 1252 |
+
{
|
| 1253 |
+
"epoch": 0.053926525109538256,
|
| 1254 |
+
"eval_loss": 3.6315455436706543,
|
| 1255 |
+
"eval_runtime": 7.6341,
|
| 1256 |
+
"eval_samples_per_second": 1247.949,
|
| 1257 |
+
"eval_steps_per_second": 0.917,
|
| 1258 |
+
"step": 3200
|
| 1259 |
+
},
|
| 1260 |
+
{
|
| 1261 |
+
"epoch": 0.05426356589147287,
|
| 1262 |
+
"grad_norm": 1.4296875,
|
| 1263 |
+
"learning_rate": 0.0003,
|
| 1264 |
+
"loss": 3.6562938690185547,
|
| 1265 |
+
"step": 3220
|
| 1266 |
+
},
|
| 1267 |
+
{
|
| 1268 |
+
"epoch": 0.054600606673407485,
|
| 1269 |
+
"grad_norm": 1.6640625,
|
| 1270 |
+
"learning_rate": 0.0003,
|
| 1271 |
+
"loss": 3.64786376953125,
|
| 1272 |
+
"step": 3240
|
| 1273 |
+
},
|
| 1274 |
+
{
|
| 1275 |
+
"epoch": 0.0549376474553421,
|
| 1276 |
+
"grad_norm": 1.890625,
|
| 1277 |
+
"learning_rate": 0.0003,
|
| 1278 |
+
"loss": 3.6485218048095702,
|
| 1279 |
+
"step": 3260
|
| 1280 |
+
},
|
| 1281 |
+
{
|
| 1282 |
+
"epoch": 0.05527468823727671,
|
| 1283 |
+
"grad_norm": 1.5,
|
| 1284 |
+
"learning_rate": 0.0003,
|
| 1285 |
+
"loss": 3.6148853302001953,
|
| 1286 |
+
"step": 3280
|
| 1287 |
+
},
|
| 1288 |
+
{
|
| 1289 |
+
"epoch": 0.055611729019211326,
|
| 1290 |
+
"grad_norm": 1.453125,
|
| 1291 |
+
"learning_rate": 0.0003,
|
| 1292 |
+
"loss": 3.6581153869628906,
|
| 1293 |
+
"step": 3300
|
| 1294 |
+
},
|
| 1295 |
+
{
|
| 1296 |
+
"epoch": 0.05594876980114594,
|
| 1297 |
+
"grad_norm": 1.5,
|
| 1298 |
+
"learning_rate": 0.0003,
|
| 1299 |
+
"loss": 3.6328590393066404,
|
| 1300 |
+
"step": 3320
|
| 1301 |
+
},
|
| 1302 |
+
{
|
| 1303 |
+
"epoch": 0.056285810583080555,
|
| 1304 |
+
"grad_norm": 1.8984375,
|
| 1305 |
+
"learning_rate": 0.0003,
|
| 1306 |
+
"loss": 3.608339309692383,
|
| 1307 |
+
"step": 3340
|
| 1308 |
+
},
|
| 1309 |
+
{
|
| 1310 |
+
"epoch": 0.056622851365015166,
|
| 1311 |
+
"grad_norm": 1.546875,
|
| 1312 |
+
"learning_rate": 0.0003,
|
| 1313 |
+
"loss": 3.6275489807128904,
|
| 1314 |
+
"step": 3360
|
| 1315 |
+
},
|
| 1316 |
+
{
|
| 1317 |
+
"epoch": 0.056959892146949784,
|
| 1318 |
+
"grad_norm": 1.4296875,
|
| 1319 |
+
"learning_rate": 0.0003,
|
| 1320 |
+
"loss": 3.5989017486572266,
|
| 1321 |
+
"step": 3380
|
| 1322 |
+
},
|
| 1323 |
+
{
|
| 1324 |
+
"epoch": 0.057296932928884395,
|
| 1325 |
+
"grad_norm": 1.2734375,
|
| 1326 |
+
"learning_rate": 0.0003,
|
| 1327 |
+
"loss": 3.6258113861083983,
|
| 1328 |
+
"step": 3400
|
| 1329 |
+
},
|
| 1330 |
+
{
|
| 1331 |
+
"epoch": 0.057296932928884395,
|
| 1332 |
+
"eval_loss": 3.6147406101226807,
|
| 1333 |
+
"eval_runtime": 7.5189,
|
| 1334 |
+
"eval_samples_per_second": 1267.079,
|
| 1335 |
+
"eval_steps_per_second": 0.931,
|
| 1336 |
+
"step": 3400
|
| 1337 |
+
}
|
| 1338 |
+
],
|
| 1339 |
+
"logging_steps": 20,
|
| 1340 |
+
"max_steps": 5000,
|
| 1341 |
+
"num_input_tokens_seen": 0,
|
| 1342 |
+
"num_train_epochs": 1,
|
| 1343 |
+
"save_steps": 100,
|
| 1344 |
+
"stateful_callbacks": {
|
| 1345 |
+
"TrainerControl": {
|
| 1346 |
+
"args": {
|
| 1347 |
+
"should_epoch_stop": false,
|
| 1348 |
+
"should_evaluate": false,
|
| 1349 |
+
"should_log": false,
|
| 1350 |
+
"should_save": true,
|
| 1351 |
+
"should_training_stop": false
|
| 1352 |
+
},
|
| 1353 |
+
"attributes": {}
|
| 1354 |
+
}
|
| 1355 |
+
},
|
| 1356 |
+
"total_flos": 82290986188800.0,
|
| 1357 |
+
"train_batch_size": 16,
|
| 1358 |
+
"trial_name": null,
|
| 1359 |
+
"trial_params": null
|
| 1360 |
+
}
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-3400/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
|
| 3 |
+
size 4920
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-400/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 2036216
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cdc56d71e15d5e56d0de962a47a3aead834194bc0b3094ae1086120c923a424a
|
| 3 |
size 2036216
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-400/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4089360
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2c19e49647f886d882d4653e3b6276951ba8cf04c7c237042f3b55328aa31393
|
| 3 |
size 4089360
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-400/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14244
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3cf9097d4513154245c48236b6ec5137b7ee2a21c9f58f2cba798ea275c6026f
|
| 3 |
size 14244
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-400/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9cd73bda6f3e39fc4545a8946a8d87b16f2d59ac45ad626e25c0e3ac41ad3e7e
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-400/trainer_state.json
CHANGED
|
@@ -2,188 +2,172 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
-
"eval_steps":
|
| 7 |
"global_step": 400,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
-
"epoch": 0.
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
-
"epoch": 0.
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate":
|
| 23 |
-
"loss": 8.
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"epoch": 0.
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate":
|
| 30 |
-
"loss": 8.
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
-
"epoch": 0.
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate":
|
| 37 |
-
"loss":
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
-
"epoch": 0.
|
| 42 |
-
"grad_norm": 1.
|
| 43 |
-
"learning_rate":
|
| 44 |
-
"loss":
|
| 45 |
-
"step": 100
|
| 46 |
-
},
|
| 47 |
-
{
|
| 48 |
-
"epoch": 0.003370407819346141,
|
| 49 |
-
"eval_loss": 7.687252998352051,
|
| 50 |
-
"eval_runtime": 7.4625,
|
| 51 |
-
"eval_samples_per_second": 1276.648,
|
| 52 |
-
"eval_steps_per_second": 0.938,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
-
"epoch": 0.
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate":
|
| 59 |
-
"loss": 7.
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
-
"epoch": 0.
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate":
|
| 66 |
-
"loss": 7.
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
-
"epoch": 0.
|
| 71 |
-
"grad_norm": 1.
|
| 72 |
-
"learning_rate":
|
| 73 |
-
"loss": 7.
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
-
"epoch": 0.
|
| 78 |
-
"grad_norm": 1.
|
| 79 |
-
"learning_rate":
|
| 80 |
-
"loss":
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
-
"epoch": 0.
|
| 85 |
-
"grad_norm": 1.
|
| 86 |
-
"learning_rate":
|
| 87 |
-
"loss":
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
-
"epoch": 0.
|
| 92 |
-
"eval_loss":
|
| 93 |
-
"eval_runtime": 7.
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
-
"epoch": 0.
|
| 100 |
-
"grad_norm": 1.
|
| 101 |
-
"learning_rate":
|
| 102 |
-
"loss":
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
-
"epoch": 0.
|
| 107 |
-
"grad_norm":
|
| 108 |
-
"learning_rate":
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
-
"epoch": 0.
|
| 114 |
-
"grad_norm":
|
| 115 |
-
"learning_rate":
|
| 116 |
-
"loss": 6.
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
-
"epoch": 0.
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate":
|
| 123 |
-
"loss":
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
-
"epoch": 0.
|
| 128 |
-
"grad_norm": 1.
|
| 129 |
-
"learning_rate":
|
| 130 |
-
"loss":
|
| 131 |
-
"step": 300
|
| 132 |
-
},
|
| 133 |
-
{
|
| 134 |
-
"epoch": 0.010111223458038422,
|
| 135 |
-
"eval_loss": 5.60944938659668,
|
| 136 |
-
"eval_runtime": 7.4311,
|
| 137 |
-
"eval_samples_per_second": 1282.036,
|
| 138 |
-
"eval_steps_per_second": 0.942,
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
-
"epoch": 0.
|
| 143 |
-
"grad_norm":
|
| 144 |
-
"learning_rate":
|
| 145 |
-
"loss":
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
-
"epoch": 0.
|
| 150 |
-
"grad_norm": 1.
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss":
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
-
"epoch": 0.
|
| 157 |
-
"grad_norm": 1.
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss":
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
-
"epoch": 0.
|
| 164 |
-
"grad_norm": 1.
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss":
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
-
"epoch": 0.
|
| 171 |
-
"grad_norm": 1.
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss":
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
-
"epoch": 0.
|
| 178 |
-
"eval_loss":
|
| 179 |
-
"eval_runtime": 7.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
}
|
| 184 |
],
|
| 185 |
"logging_steps": 20,
|
| 186 |
-
"max_steps":
|
| 187 |
"num_input_tokens_seen": 0,
|
| 188 |
"num_train_epochs": 1,
|
| 189 |
"save_steps": 100,
|
|
@@ -199,8 +183,8 @@
|
|
| 199 |
"attributes": {}
|
| 200 |
}
|
| 201 |
},
|
| 202 |
-
"total_flos":
|
| 203 |
-
"train_batch_size":
|
| 204 |
"trial_name": null,
|
| 205 |
"trial_params": null
|
| 206 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.006740815638692282,
|
| 6 |
+
"eval_steps": 200,
|
| 7 |
"global_step": 400,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
+
"epoch": 0.0003370407819346141,
|
| 14 |
+
"grad_norm": 1.4140625,
|
| 15 |
+
"learning_rate": 5.7e-06,
|
| 16 |
+
"loss": 8.324227142333985,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
+
"epoch": 0.0006740815638692282,
|
| 21 |
+
"grad_norm": 1.5390625,
|
| 22 |
+
"learning_rate": 1.17e-05,
|
| 23 |
+
"loss": 8.318060302734375,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
+
"epoch": 0.0010111223458038423,
|
| 28 |
+
"grad_norm": 1.609375,
|
| 29 |
+
"learning_rate": 1.7699999999999997e-05,
|
| 30 |
+
"loss": 8.293100738525391,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
+
"epoch": 0.0013481631277384564,
|
| 35 |
+
"grad_norm": 1.7578125,
|
| 36 |
+
"learning_rate": 2.3699999999999997e-05,
|
| 37 |
+
"loss": 8.227334594726562,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
+
"epoch": 0.0016852039096730705,
|
| 42 |
+
"grad_norm": 1.5234375,
|
| 43 |
+
"learning_rate": 2.97e-05,
|
| 44 |
+
"loss": 8.111893463134766,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
+
"epoch": 0.0020222446916076846,
|
| 49 |
+
"grad_norm": 1.3046875,
|
| 50 |
+
"learning_rate": 3.5699999999999994e-05,
|
| 51 |
+
"loss": 7.976696014404297,
|
| 52 |
"step": 120
|
| 53 |
},
|
| 54 |
{
|
| 55 |
+
"epoch": 0.0023592854735422987,
|
| 56 |
+
"grad_norm": 1.296875,
|
| 57 |
+
"learning_rate": 4.17e-05,
|
| 58 |
+
"loss": 7.851339721679688,
|
| 59 |
"step": 140
|
| 60 |
},
|
| 61 |
{
|
| 62 |
+
"epoch": 0.002696326255476913,
|
| 63 |
+
"grad_norm": 1.3046875,
|
| 64 |
+
"learning_rate": 4.7699999999999994e-05,
|
| 65 |
+
"loss": 7.724923706054687,
|
| 66 |
"step": 160
|
| 67 |
},
|
| 68 |
{
|
| 69 |
+
"epoch": 0.003033367037411527,
|
| 70 |
+
"grad_norm": 1.3125,
|
| 71 |
+
"learning_rate": 5.369999999999999e-05,
|
| 72 |
+
"loss": 7.58428726196289,
|
| 73 |
"step": 180
|
| 74 |
},
|
| 75 |
{
|
| 76 |
+
"epoch": 0.003370407819346141,
|
| 77 |
+
"grad_norm": 1.25,
|
| 78 |
+
"learning_rate": 5.97e-05,
|
| 79 |
+
"loss": 7.439914703369141,
|
| 80 |
"step": 200
|
| 81 |
},
|
| 82 |
{
|
| 83 |
+
"epoch": 0.003370407819346141,
|
| 84 |
+
"eval_loss": 7.356490135192871,
|
| 85 |
+
"eval_runtime": 7.5119,
|
| 86 |
+
"eval_samples_per_second": 1268.257,
|
| 87 |
+
"eval_steps_per_second": 0.932,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
+
"epoch": 0.003707448601280755,
|
| 92 |
+
"grad_norm": 1.25,
|
| 93 |
+
"learning_rate": 6.57e-05,
|
| 94 |
+
"loss": 7.283377075195313,
|
| 95 |
"step": 220
|
| 96 |
},
|
| 97 |
{
|
| 98 |
+
"epoch": 0.004044489383215369,
|
| 99 |
+
"grad_norm": 1.25,
|
| 100 |
+
"learning_rate": 7.17e-05,
|
| 101 |
+
"loss": 7.127851104736328,
|
| 102 |
"step": 240
|
| 103 |
},
|
| 104 |
{
|
| 105 |
+
"epoch": 0.004381530165149983,
|
| 106 |
+
"grad_norm": 1.2109375,
|
| 107 |
+
"learning_rate": 7.769999999999999e-05,
|
| 108 |
+
"loss": 6.964313507080078,
|
| 109 |
"step": 260
|
| 110 |
},
|
| 111 |
{
|
| 112 |
+
"epoch": 0.0047185709470845974,
|
| 113 |
+
"grad_norm": 1.171875,
|
| 114 |
+
"learning_rate": 8.37e-05,
|
| 115 |
+
"loss": 6.807338714599609,
|
| 116 |
"step": 280
|
| 117 |
},
|
| 118 |
{
|
| 119 |
+
"epoch": 0.005055611729019211,
|
| 120 |
+
"grad_norm": 1.1328125,
|
| 121 |
+
"learning_rate": 8.969999999999998e-05,
|
| 122 |
+
"loss": 6.667655181884766,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 123 |
"step": 300
|
| 124 |
},
|
| 125 |
{
|
| 126 |
+
"epoch": 0.005392652510953826,
|
| 127 |
+
"grad_norm": 1.109375,
|
| 128 |
+
"learning_rate": 9.57e-05,
|
| 129 |
+
"loss": 6.523377227783203,
|
| 130 |
"step": 320
|
| 131 |
},
|
| 132 |
{
|
| 133 |
+
"epoch": 0.005729693292888439,
|
| 134 |
+
"grad_norm": 1.1171875,
|
| 135 |
+
"learning_rate": 0.00010169999999999999,
|
| 136 |
+
"loss": 6.383005142211914,
|
| 137 |
"step": 340
|
| 138 |
},
|
| 139 |
{
|
| 140 |
+
"epoch": 0.006066734074823054,
|
| 141 |
+
"grad_norm": 1.6875,
|
| 142 |
+
"learning_rate": 0.00010769999999999999,
|
| 143 |
+
"loss": 6.261091232299805,
|
| 144 |
"step": 360
|
| 145 |
},
|
| 146 |
{
|
| 147 |
+
"epoch": 0.0064037748567576675,
|
| 148 |
+
"grad_norm": 1.140625,
|
| 149 |
+
"learning_rate": 0.00011369999999999999,
|
| 150 |
+
"loss": 6.122833251953125,
|
| 151 |
"step": 380
|
| 152 |
},
|
| 153 |
{
|
| 154 |
+
"epoch": 0.006740815638692282,
|
| 155 |
+
"grad_norm": 1.3984375,
|
| 156 |
+
"learning_rate": 0.0001197,
|
| 157 |
+
"loss": 6.019657897949219,
|
| 158 |
"step": 400
|
| 159 |
},
|
| 160 |
{
|
| 161 |
+
"epoch": 0.006740815638692282,
|
| 162 |
+
"eval_loss": 5.966014385223389,
|
| 163 |
+
"eval_runtime": 7.516,
|
| 164 |
+
"eval_samples_per_second": 1267.556,
|
| 165 |
+
"eval_steps_per_second": 0.931,
|
| 166 |
"step": 400
|
| 167 |
}
|
| 168 |
],
|
| 169 |
"logging_steps": 20,
|
| 170 |
+
"max_steps": 5000,
|
| 171 |
"num_input_tokens_seen": 0,
|
| 172 |
"num_train_epochs": 1,
|
| 173 |
"save_steps": 100,
|
|
|
|
| 183 |
"attributes": {}
|
| 184 |
}
|
| 185 |
},
|
| 186 |
+
"total_flos": 9681292492800.0,
|
| 187 |
+
"train_batch_size": 16,
|
| 188 |
"trial_name": null,
|
| 189 |
"trial_params": null
|
| 190 |
}
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-400/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-500/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 2036216
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8268e70c42413ec7b95cd060155f41891df1ef6759c56990157f3f55c74f345a
|
| 3 |
size 2036216
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-500/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4089360
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fa6f3c8b6645828936657c806f68d113379205778239766bd5d55e26ef1fda85
|
| 3 |
size 4089360
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-500/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14244
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3cf9097d4513154245c48236b6ec5137b7ee2a21c9f58f2cba798ea275c6026f
|
| 3 |
size 14244
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-500/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2627ccec0bb9a51b7d9d753a9441035aec88f305994eb3b5ccbb3e0571f519d6
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-500/trainer_state.json
CHANGED
|
@@ -2,231 +2,207 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
-
"eval_steps":
|
| 7 |
"global_step": 500,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
-
"epoch": 0.
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
-
"epoch": 0.
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate":
|
| 23 |
-
"loss": 8.
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"epoch": 0.
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate":
|
| 30 |
-
"loss": 8.
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
-
"epoch": 0.
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate":
|
| 37 |
-
"loss":
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
-
"epoch": 0.
|
| 42 |
-
"grad_norm": 1.
|
| 43 |
-
"learning_rate":
|
| 44 |
-
"loss":
|
| 45 |
-
"step": 100
|
| 46 |
-
},
|
| 47 |
-
{
|
| 48 |
-
"epoch": 0.003370407819346141,
|
| 49 |
-
"eval_loss": 7.687252998352051,
|
| 50 |
-
"eval_runtime": 7.4625,
|
| 51 |
-
"eval_samples_per_second": 1276.648,
|
| 52 |
-
"eval_steps_per_second": 0.938,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
-
"epoch": 0.
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate":
|
| 59 |
-
"loss": 7.
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
-
"epoch": 0.
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate":
|
| 66 |
-
"loss": 7.
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
-
"epoch": 0.
|
| 71 |
-
"grad_norm": 1.
|
| 72 |
-
"learning_rate":
|
| 73 |
-
"loss": 7.
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
-
"epoch": 0.
|
| 78 |
-
"grad_norm": 1.
|
| 79 |
-
"learning_rate":
|
| 80 |
-
"loss":
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
-
"epoch": 0.
|
| 85 |
-
"grad_norm": 1.
|
| 86 |
-
"learning_rate":
|
| 87 |
-
"loss":
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
-
"epoch": 0.
|
| 92 |
-
"eval_loss":
|
| 93 |
-
"eval_runtime": 7.
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
-
"epoch": 0.
|
| 100 |
-
"grad_norm": 1.
|
| 101 |
-
"learning_rate":
|
| 102 |
-
"loss":
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
-
"epoch": 0.
|
| 107 |
-
"grad_norm":
|
| 108 |
-
"learning_rate":
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
-
"epoch": 0.
|
| 114 |
-
"grad_norm":
|
| 115 |
-
"learning_rate":
|
| 116 |
-
"loss": 6.
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
-
"epoch": 0.
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate":
|
| 123 |
-
"loss":
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
-
"epoch": 0.
|
| 128 |
-
"grad_norm": 1.
|
| 129 |
-
"learning_rate":
|
| 130 |
-
"loss":
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
-
"epoch": 0.
|
| 135 |
-
"
|
| 136 |
-
"
|
| 137 |
-
"
|
| 138 |
-
"eval_steps_per_second": 0.942,
|
| 139 |
-
"step": 300
|
| 140 |
-
},
|
| 141 |
-
{
|
| 142 |
-
"epoch": 0.010785305021907651,
|
| 143 |
-
"grad_norm": 0.9375,
|
| 144 |
-
"learning_rate": 0.0001914,
|
| 145 |
-
"loss": 5.544954299926758,
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
-
"epoch": 0.
|
| 150 |
-
"grad_norm": 1.
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss":
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
-
"epoch": 0.
|
| 157 |
-
"grad_norm": 1.
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss":
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
-
"epoch": 0.
|
| 164 |
-
"grad_norm": 1.
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss":
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
-
"epoch": 0.
|
| 171 |
-
"grad_norm": 1.
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss":
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
-
"epoch": 0.
|
| 178 |
-
"eval_loss":
|
| 179 |
-
"eval_runtime": 7.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
-
"epoch": 0.
|
| 186 |
-
"grad_norm": 0.
|
| 187 |
-
"learning_rate": 0.
|
| 188 |
-
"loss":
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
-
"epoch": 0.
|
| 193 |
-
"grad_norm": 1.
|
| 194 |
-
"learning_rate": 0.
|
| 195 |
-
"loss":
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
-
"epoch": 0.
|
| 200 |
-
"grad_norm":
|
| 201 |
-
"learning_rate": 0.
|
| 202 |
-
"loss":
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
-
"epoch": 0.
|
| 207 |
-
"grad_norm":
|
| 208 |
-
"learning_rate": 0.
|
| 209 |
-
"loss":
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
-
"epoch": 0.
|
| 214 |
-
"grad_norm": 1.
|
| 215 |
-
"learning_rate": 0.
|
| 216 |
-
"loss":
|
| 217 |
-
"step": 500
|
| 218 |
-
},
|
| 219 |
-
{
|
| 220 |
-
"epoch": 0.016852039096730706,
|
| 221 |
-
"eval_loss": 4.591919898986816,
|
| 222 |
-
"eval_runtime": 7.3895,
|
| 223 |
-
"eval_samples_per_second": 1289.263,
|
| 224 |
-
"eval_steps_per_second": 0.947,
|
| 225 |
"step": 500
|
| 226 |
}
|
| 227 |
],
|
| 228 |
"logging_steps": 20,
|
| 229 |
-
"max_steps":
|
| 230 |
"num_input_tokens_seen": 0,
|
| 231 |
"num_train_epochs": 1,
|
| 232 |
"save_steps": 100,
|
|
@@ -242,8 +218,8 @@
|
|
| 242 |
"attributes": {}
|
| 243 |
}
|
| 244 |
},
|
| 245 |
-
"total_flos":
|
| 246 |
-
"train_batch_size":
|
| 247 |
"trial_name": null,
|
| 248 |
"trial_params": null
|
| 249 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.008426019548365353,
|
| 6 |
+
"eval_steps": 200,
|
| 7 |
"global_step": 500,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
+
"epoch": 0.0003370407819346141,
|
| 14 |
+
"grad_norm": 1.4140625,
|
| 15 |
+
"learning_rate": 5.7e-06,
|
| 16 |
+
"loss": 8.324227142333985,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
+
"epoch": 0.0006740815638692282,
|
| 21 |
+
"grad_norm": 1.5390625,
|
| 22 |
+
"learning_rate": 1.17e-05,
|
| 23 |
+
"loss": 8.318060302734375,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
+
"epoch": 0.0010111223458038423,
|
| 28 |
+
"grad_norm": 1.609375,
|
| 29 |
+
"learning_rate": 1.7699999999999997e-05,
|
| 30 |
+
"loss": 8.293100738525391,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
+
"epoch": 0.0013481631277384564,
|
| 35 |
+
"grad_norm": 1.7578125,
|
| 36 |
+
"learning_rate": 2.3699999999999997e-05,
|
| 37 |
+
"loss": 8.227334594726562,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
+
"epoch": 0.0016852039096730705,
|
| 42 |
+
"grad_norm": 1.5234375,
|
| 43 |
+
"learning_rate": 2.97e-05,
|
| 44 |
+
"loss": 8.111893463134766,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
+
"epoch": 0.0020222446916076846,
|
| 49 |
+
"grad_norm": 1.3046875,
|
| 50 |
+
"learning_rate": 3.5699999999999994e-05,
|
| 51 |
+
"loss": 7.976696014404297,
|
| 52 |
"step": 120
|
| 53 |
},
|
| 54 |
{
|
| 55 |
+
"epoch": 0.0023592854735422987,
|
| 56 |
+
"grad_norm": 1.296875,
|
| 57 |
+
"learning_rate": 4.17e-05,
|
| 58 |
+
"loss": 7.851339721679688,
|
| 59 |
"step": 140
|
| 60 |
},
|
| 61 |
{
|
| 62 |
+
"epoch": 0.002696326255476913,
|
| 63 |
+
"grad_norm": 1.3046875,
|
| 64 |
+
"learning_rate": 4.7699999999999994e-05,
|
| 65 |
+
"loss": 7.724923706054687,
|
| 66 |
"step": 160
|
| 67 |
},
|
| 68 |
{
|
| 69 |
+
"epoch": 0.003033367037411527,
|
| 70 |
+
"grad_norm": 1.3125,
|
| 71 |
+
"learning_rate": 5.369999999999999e-05,
|
| 72 |
+
"loss": 7.58428726196289,
|
| 73 |
"step": 180
|
| 74 |
},
|
| 75 |
{
|
| 76 |
+
"epoch": 0.003370407819346141,
|
| 77 |
+
"grad_norm": 1.25,
|
| 78 |
+
"learning_rate": 5.97e-05,
|
| 79 |
+
"loss": 7.439914703369141,
|
| 80 |
"step": 200
|
| 81 |
},
|
| 82 |
{
|
| 83 |
+
"epoch": 0.003370407819346141,
|
| 84 |
+
"eval_loss": 7.356490135192871,
|
| 85 |
+
"eval_runtime": 7.5119,
|
| 86 |
+
"eval_samples_per_second": 1268.257,
|
| 87 |
+
"eval_steps_per_second": 0.932,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
+
"epoch": 0.003707448601280755,
|
| 92 |
+
"grad_norm": 1.25,
|
| 93 |
+
"learning_rate": 6.57e-05,
|
| 94 |
+
"loss": 7.283377075195313,
|
| 95 |
"step": 220
|
| 96 |
},
|
| 97 |
{
|
| 98 |
+
"epoch": 0.004044489383215369,
|
| 99 |
+
"grad_norm": 1.25,
|
| 100 |
+
"learning_rate": 7.17e-05,
|
| 101 |
+
"loss": 7.127851104736328,
|
| 102 |
"step": 240
|
| 103 |
},
|
| 104 |
{
|
| 105 |
+
"epoch": 0.004381530165149983,
|
| 106 |
+
"grad_norm": 1.2109375,
|
| 107 |
+
"learning_rate": 7.769999999999999e-05,
|
| 108 |
+
"loss": 6.964313507080078,
|
| 109 |
"step": 260
|
| 110 |
},
|
| 111 |
{
|
| 112 |
+
"epoch": 0.0047185709470845974,
|
| 113 |
+
"grad_norm": 1.171875,
|
| 114 |
+
"learning_rate": 8.37e-05,
|
| 115 |
+
"loss": 6.807338714599609,
|
| 116 |
"step": 280
|
| 117 |
},
|
| 118 |
{
|
| 119 |
+
"epoch": 0.005055611729019211,
|
| 120 |
+
"grad_norm": 1.1328125,
|
| 121 |
+
"learning_rate": 8.969999999999998e-05,
|
| 122 |
+
"loss": 6.667655181884766,
|
| 123 |
"step": 300
|
| 124 |
},
|
| 125 |
{
|
| 126 |
+
"epoch": 0.005392652510953826,
|
| 127 |
+
"grad_norm": 1.109375,
|
| 128 |
+
"learning_rate": 9.57e-05,
|
| 129 |
+
"loss": 6.523377227783203,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 130 |
"step": 320
|
| 131 |
},
|
| 132 |
{
|
| 133 |
+
"epoch": 0.005729693292888439,
|
| 134 |
+
"grad_norm": 1.1171875,
|
| 135 |
+
"learning_rate": 0.00010169999999999999,
|
| 136 |
+
"loss": 6.383005142211914,
|
| 137 |
"step": 340
|
| 138 |
},
|
| 139 |
{
|
| 140 |
+
"epoch": 0.006066734074823054,
|
| 141 |
+
"grad_norm": 1.6875,
|
| 142 |
+
"learning_rate": 0.00010769999999999999,
|
| 143 |
+
"loss": 6.261091232299805,
|
| 144 |
"step": 360
|
| 145 |
},
|
| 146 |
{
|
| 147 |
+
"epoch": 0.0064037748567576675,
|
| 148 |
+
"grad_norm": 1.140625,
|
| 149 |
+
"learning_rate": 0.00011369999999999999,
|
| 150 |
+
"loss": 6.122833251953125,
|
| 151 |
"step": 380
|
| 152 |
},
|
| 153 |
{
|
| 154 |
+
"epoch": 0.006740815638692282,
|
| 155 |
+
"grad_norm": 1.3984375,
|
| 156 |
+
"learning_rate": 0.0001197,
|
| 157 |
+
"loss": 6.019657897949219,
|
| 158 |
"step": 400
|
| 159 |
},
|
| 160 |
{
|
| 161 |
+
"epoch": 0.006740815638692282,
|
| 162 |
+
"eval_loss": 5.966014385223389,
|
| 163 |
+
"eval_runtime": 7.516,
|
| 164 |
+
"eval_samples_per_second": 1267.556,
|
| 165 |
+
"eval_steps_per_second": 0.931,
|
| 166 |
"step": 400
|
| 167 |
},
|
| 168 |
{
|
| 169 |
+
"epoch": 0.007077856420626896,
|
| 170 |
+
"grad_norm": 0.98046875,
|
| 171 |
+
"learning_rate": 0.0001257,
|
| 172 |
+
"loss": 5.9373779296875,
|
| 173 |
"step": 420
|
| 174 |
},
|
| 175 |
{
|
| 176 |
+
"epoch": 0.00741489720256151,
|
| 177 |
+
"grad_norm": 1.6328125,
|
| 178 |
+
"learning_rate": 0.00013169999999999998,
|
| 179 |
+
"loss": 5.839211273193359,
|
| 180 |
"step": 440
|
| 181 |
},
|
| 182 |
{
|
| 183 |
+
"epoch": 0.007751937984496124,
|
| 184 |
+
"grad_norm": 0.9609375,
|
| 185 |
+
"learning_rate": 0.00013769999999999999,
|
| 186 |
+
"loss": 5.740922927856445,
|
| 187 |
"step": 460
|
| 188 |
},
|
| 189 |
{
|
| 190 |
+
"epoch": 0.008088978766430738,
|
| 191 |
+
"grad_norm": 0.90234375,
|
| 192 |
+
"learning_rate": 0.00014369999999999997,
|
| 193 |
+
"loss": 5.6399181365966795,
|
| 194 |
"step": 480
|
| 195 |
},
|
| 196 |
{
|
| 197 |
+
"epoch": 0.008426019548365353,
|
| 198 |
+
"grad_norm": 1.1796875,
|
| 199 |
+
"learning_rate": 0.00014969999999999998,
|
| 200 |
+
"loss": 5.560699081420898,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 201 |
"step": 500
|
| 202 |
}
|
| 203 |
],
|
| 204 |
"logging_steps": 20,
|
| 205 |
+
"max_steps": 5000,
|
| 206 |
"num_input_tokens_seen": 0,
|
| 207 |
"num_train_epochs": 1,
|
| 208 |
"save_steps": 100,
|
|
|
|
| 218 |
"attributes": {}
|
| 219 |
}
|
| 220 |
},
|
| 221 |
+
"total_flos": 12101615616000.0,
|
| 222 |
+
"train_batch_size": 16,
|
| 223 |
"trial_name": null,
|
| 224 |
"trial_params": null
|
| 225 |
}
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-500/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-600/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 2036216
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1ab2e2cfacb38b4d1367fa6d371a767c6bb149e9ca0f5bc2e0a9bd8afc9ba2f9
|
| 3 |
size 2036216
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-600/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4089360
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:973f768bdaf58bb9c10b90ad211f2d0f7fb6bec95950be6f045639f355cb72cb
|
| 3 |
size 4089360
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-600/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14244
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f37c40ce327861a7ca13b719d3aa37510a143368b6e74358bdb14becb3899e1e
|
| 3 |
size 14244
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-600/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:44937bb66e00fa484302cf89e685ffa903e6b6f2eac5fdae0a7e91e0560ce411
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-600/trainer_state.json
CHANGED
|
@@ -2,274 +2,250 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
-
"eval_steps":
|
| 7 |
"global_step": 600,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
-
"epoch": 0.
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
-
"epoch": 0.
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate":
|
| 23 |
-
"loss": 8.
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"epoch": 0.
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate":
|
| 30 |
-
"loss": 8.
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
-
"epoch": 0.
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate":
|
| 37 |
-
"loss":
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
-
"epoch": 0.
|
| 42 |
-
"grad_norm": 1.
|
| 43 |
-
"learning_rate":
|
| 44 |
-
"loss":
|
| 45 |
-
"step": 100
|
| 46 |
-
},
|
| 47 |
-
{
|
| 48 |
-
"epoch": 0.003370407819346141,
|
| 49 |
-
"eval_loss": 7.687252998352051,
|
| 50 |
-
"eval_runtime": 7.4625,
|
| 51 |
-
"eval_samples_per_second": 1276.648,
|
| 52 |
-
"eval_steps_per_second": 0.938,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
-
"epoch": 0.
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate":
|
| 59 |
-
"loss": 7.
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
-
"epoch": 0.
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate":
|
| 66 |
-
"loss": 7.
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
-
"epoch": 0.
|
| 71 |
-
"grad_norm": 1.
|
| 72 |
-
"learning_rate":
|
| 73 |
-
"loss": 7.
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
-
"epoch": 0.
|
| 78 |
-
"grad_norm": 1.
|
| 79 |
-
"learning_rate":
|
| 80 |
-
"loss":
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
-
"epoch": 0.
|
| 85 |
-
"grad_norm": 1.
|
| 86 |
-
"learning_rate":
|
| 87 |
-
"loss":
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
-
"epoch": 0.
|
| 92 |
-
"eval_loss":
|
| 93 |
-
"eval_runtime": 7.
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
-
"epoch": 0.
|
| 100 |
-
"grad_norm": 1.
|
| 101 |
-
"learning_rate":
|
| 102 |
-
"loss":
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
-
"epoch": 0.
|
| 107 |
-
"grad_norm":
|
| 108 |
-
"learning_rate":
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
-
"epoch": 0.
|
| 114 |
-
"grad_norm":
|
| 115 |
-
"learning_rate":
|
| 116 |
-
"loss": 6.
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
-
"epoch": 0.
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate":
|
| 123 |
-
"loss":
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
-
"epoch": 0.
|
| 128 |
-
"grad_norm": 1.
|
| 129 |
-
"learning_rate":
|
| 130 |
-
"loss":
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
-
"epoch": 0.
|
| 135 |
-
"
|
| 136 |
-
"
|
| 137 |
-
"
|
| 138 |
-
"eval_steps_per_second": 0.942,
|
| 139 |
-
"step": 300
|
| 140 |
-
},
|
| 141 |
-
{
|
| 142 |
-
"epoch": 0.010785305021907651,
|
| 143 |
-
"grad_norm": 0.9375,
|
| 144 |
-
"learning_rate": 0.0001914,
|
| 145 |
-
"loss": 5.544954299926758,
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
-
"epoch": 0.
|
| 150 |
-
"grad_norm": 1.
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss":
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
-
"epoch": 0.
|
| 157 |
-
"grad_norm": 1.
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss":
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
-
"epoch": 0.
|
| 164 |
-
"grad_norm": 1.
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss":
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
-
"epoch": 0.
|
| 171 |
-
"grad_norm": 1.
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss":
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
-
"epoch": 0.
|
| 178 |
-
"eval_loss":
|
| 179 |
-
"eval_runtime": 7.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
-
"epoch": 0.
|
| 186 |
-
"grad_norm": 0.
|
| 187 |
-
"learning_rate": 0.
|
| 188 |
-
"loss":
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
-
"epoch": 0.
|
| 193 |
-
"grad_norm": 1.
|
| 194 |
-
"learning_rate": 0.
|
| 195 |
-
"loss":
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
-
"epoch": 0.
|
| 200 |
-
"grad_norm":
|
| 201 |
-
"learning_rate": 0.
|
| 202 |
-
"loss":
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
-
"epoch": 0.
|
| 207 |
-
"grad_norm":
|
| 208 |
-
"learning_rate": 0.
|
| 209 |
-
"loss":
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
-
"epoch": 0.
|
| 214 |
-
"grad_norm": 1.
|
| 215 |
-
"learning_rate": 0.
|
| 216 |
-
"loss":
|
| 217 |
-
"step": 500
|
| 218 |
-
},
|
| 219 |
-
{
|
| 220 |
-
"epoch": 0.016852039096730706,
|
| 221 |
-
"eval_loss": 4.591919898986816,
|
| 222 |
-
"eval_runtime": 7.3895,
|
| 223 |
-
"eval_samples_per_second": 1289.263,
|
| 224 |
-
"eval_steps_per_second": 0.947,
|
| 225 |
"step": 500
|
| 226 |
},
|
| 227 |
{
|
| 228 |
-
"epoch": 0.
|
| 229 |
-
"grad_norm":
|
| 230 |
-
"learning_rate": 0.
|
| 231 |
-
"loss":
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
-
"epoch": 0.
|
| 236 |
-
"grad_norm": 1.
|
| 237 |
-
"learning_rate": 0.
|
| 238 |
-
"loss":
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
-
"epoch": 0.
|
| 243 |
-
"grad_norm": 1.
|
| 244 |
-
"learning_rate": 0.
|
| 245 |
-
"loss":
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
-
"epoch": 0.
|
| 250 |
-
"grad_norm": 1.
|
| 251 |
-
"learning_rate": 0.
|
| 252 |
-
"loss":
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
-
"epoch": 0.
|
| 257 |
-
"grad_norm":
|
| 258 |
-
"learning_rate": 0.
|
| 259 |
-
"loss":
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
-
"epoch": 0.
|
| 264 |
-
"eval_loss":
|
| 265 |
-
"eval_runtime": 7.
|
| 266 |
-
"eval_samples_per_second":
|
| 267 |
-
"eval_steps_per_second": 0.
|
| 268 |
"step": 600
|
| 269 |
}
|
| 270 |
],
|
| 271 |
"logging_steps": 20,
|
| 272 |
-
"max_steps":
|
| 273 |
"num_input_tokens_seen": 0,
|
| 274 |
"num_train_epochs": 1,
|
| 275 |
"save_steps": 100,
|
|
@@ -285,8 +261,8 @@
|
|
| 285 |
"attributes": {}
|
| 286 |
}
|
| 287 |
},
|
| 288 |
-
"total_flos":
|
| 289 |
-
"train_batch_size":
|
| 290 |
"trial_name": null,
|
| 291 |
"trial_params": null
|
| 292 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.010111223458038422,
|
| 6 |
+
"eval_steps": 200,
|
| 7 |
"global_step": 600,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
+
"epoch": 0.0003370407819346141,
|
| 14 |
+
"grad_norm": 1.4140625,
|
| 15 |
+
"learning_rate": 5.7e-06,
|
| 16 |
+
"loss": 8.324227142333985,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
+
"epoch": 0.0006740815638692282,
|
| 21 |
+
"grad_norm": 1.5390625,
|
| 22 |
+
"learning_rate": 1.17e-05,
|
| 23 |
+
"loss": 8.318060302734375,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
+
"epoch": 0.0010111223458038423,
|
| 28 |
+
"grad_norm": 1.609375,
|
| 29 |
+
"learning_rate": 1.7699999999999997e-05,
|
| 30 |
+
"loss": 8.293100738525391,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
+
"epoch": 0.0013481631277384564,
|
| 35 |
+
"grad_norm": 1.7578125,
|
| 36 |
+
"learning_rate": 2.3699999999999997e-05,
|
| 37 |
+
"loss": 8.227334594726562,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
+
"epoch": 0.0016852039096730705,
|
| 42 |
+
"grad_norm": 1.5234375,
|
| 43 |
+
"learning_rate": 2.97e-05,
|
| 44 |
+
"loss": 8.111893463134766,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
+
"epoch": 0.0020222446916076846,
|
| 49 |
+
"grad_norm": 1.3046875,
|
| 50 |
+
"learning_rate": 3.5699999999999994e-05,
|
| 51 |
+
"loss": 7.976696014404297,
|
| 52 |
"step": 120
|
| 53 |
},
|
| 54 |
{
|
| 55 |
+
"epoch": 0.0023592854735422987,
|
| 56 |
+
"grad_norm": 1.296875,
|
| 57 |
+
"learning_rate": 4.17e-05,
|
| 58 |
+
"loss": 7.851339721679688,
|
| 59 |
"step": 140
|
| 60 |
},
|
| 61 |
{
|
| 62 |
+
"epoch": 0.002696326255476913,
|
| 63 |
+
"grad_norm": 1.3046875,
|
| 64 |
+
"learning_rate": 4.7699999999999994e-05,
|
| 65 |
+
"loss": 7.724923706054687,
|
| 66 |
"step": 160
|
| 67 |
},
|
| 68 |
{
|
| 69 |
+
"epoch": 0.003033367037411527,
|
| 70 |
+
"grad_norm": 1.3125,
|
| 71 |
+
"learning_rate": 5.369999999999999e-05,
|
| 72 |
+
"loss": 7.58428726196289,
|
| 73 |
"step": 180
|
| 74 |
},
|
| 75 |
{
|
| 76 |
+
"epoch": 0.003370407819346141,
|
| 77 |
+
"grad_norm": 1.25,
|
| 78 |
+
"learning_rate": 5.97e-05,
|
| 79 |
+
"loss": 7.439914703369141,
|
| 80 |
"step": 200
|
| 81 |
},
|
| 82 |
{
|
| 83 |
+
"epoch": 0.003370407819346141,
|
| 84 |
+
"eval_loss": 7.356490135192871,
|
| 85 |
+
"eval_runtime": 7.5119,
|
| 86 |
+
"eval_samples_per_second": 1268.257,
|
| 87 |
+
"eval_steps_per_second": 0.932,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
+
"epoch": 0.003707448601280755,
|
| 92 |
+
"grad_norm": 1.25,
|
| 93 |
+
"learning_rate": 6.57e-05,
|
| 94 |
+
"loss": 7.283377075195313,
|
| 95 |
"step": 220
|
| 96 |
},
|
| 97 |
{
|
| 98 |
+
"epoch": 0.004044489383215369,
|
| 99 |
+
"grad_norm": 1.25,
|
| 100 |
+
"learning_rate": 7.17e-05,
|
| 101 |
+
"loss": 7.127851104736328,
|
| 102 |
"step": 240
|
| 103 |
},
|
| 104 |
{
|
| 105 |
+
"epoch": 0.004381530165149983,
|
| 106 |
+
"grad_norm": 1.2109375,
|
| 107 |
+
"learning_rate": 7.769999999999999e-05,
|
| 108 |
+
"loss": 6.964313507080078,
|
| 109 |
"step": 260
|
| 110 |
},
|
| 111 |
{
|
| 112 |
+
"epoch": 0.0047185709470845974,
|
| 113 |
+
"grad_norm": 1.171875,
|
| 114 |
+
"learning_rate": 8.37e-05,
|
| 115 |
+
"loss": 6.807338714599609,
|
| 116 |
"step": 280
|
| 117 |
},
|
| 118 |
{
|
| 119 |
+
"epoch": 0.005055611729019211,
|
| 120 |
+
"grad_norm": 1.1328125,
|
| 121 |
+
"learning_rate": 8.969999999999998e-05,
|
| 122 |
+
"loss": 6.667655181884766,
|
| 123 |
"step": 300
|
| 124 |
},
|
| 125 |
{
|
| 126 |
+
"epoch": 0.005392652510953826,
|
| 127 |
+
"grad_norm": 1.109375,
|
| 128 |
+
"learning_rate": 9.57e-05,
|
| 129 |
+
"loss": 6.523377227783203,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 130 |
"step": 320
|
| 131 |
},
|
| 132 |
{
|
| 133 |
+
"epoch": 0.005729693292888439,
|
| 134 |
+
"grad_norm": 1.1171875,
|
| 135 |
+
"learning_rate": 0.00010169999999999999,
|
| 136 |
+
"loss": 6.383005142211914,
|
| 137 |
"step": 340
|
| 138 |
},
|
| 139 |
{
|
| 140 |
+
"epoch": 0.006066734074823054,
|
| 141 |
+
"grad_norm": 1.6875,
|
| 142 |
+
"learning_rate": 0.00010769999999999999,
|
| 143 |
+
"loss": 6.261091232299805,
|
| 144 |
"step": 360
|
| 145 |
},
|
| 146 |
{
|
| 147 |
+
"epoch": 0.0064037748567576675,
|
| 148 |
+
"grad_norm": 1.140625,
|
| 149 |
+
"learning_rate": 0.00011369999999999999,
|
| 150 |
+
"loss": 6.122833251953125,
|
| 151 |
"step": 380
|
| 152 |
},
|
| 153 |
{
|
| 154 |
+
"epoch": 0.006740815638692282,
|
| 155 |
+
"grad_norm": 1.3984375,
|
| 156 |
+
"learning_rate": 0.0001197,
|
| 157 |
+
"loss": 6.019657897949219,
|
| 158 |
"step": 400
|
| 159 |
},
|
| 160 |
{
|
| 161 |
+
"epoch": 0.006740815638692282,
|
| 162 |
+
"eval_loss": 5.966014385223389,
|
| 163 |
+
"eval_runtime": 7.516,
|
| 164 |
+
"eval_samples_per_second": 1267.556,
|
| 165 |
+
"eval_steps_per_second": 0.931,
|
| 166 |
"step": 400
|
| 167 |
},
|
| 168 |
{
|
| 169 |
+
"epoch": 0.007077856420626896,
|
| 170 |
+
"grad_norm": 0.98046875,
|
| 171 |
+
"learning_rate": 0.0001257,
|
| 172 |
+
"loss": 5.9373779296875,
|
| 173 |
"step": 420
|
| 174 |
},
|
| 175 |
{
|
| 176 |
+
"epoch": 0.00741489720256151,
|
| 177 |
+
"grad_norm": 1.6328125,
|
| 178 |
+
"learning_rate": 0.00013169999999999998,
|
| 179 |
+
"loss": 5.839211273193359,
|
| 180 |
"step": 440
|
| 181 |
},
|
| 182 |
{
|
| 183 |
+
"epoch": 0.007751937984496124,
|
| 184 |
+
"grad_norm": 0.9609375,
|
| 185 |
+
"learning_rate": 0.00013769999999999999,
|
| 186 |
+
"loss": 5.740922927856445,
|
| 187 |
"step": 460
|
| 188 |
},
|
| 189 |
{
|
| 190 |
+
"epoch": 0.008088978766430738,
|
| 191 |
+
"grad_norm": 0.90234375,
|
| 192 |
+
"learning_rate": 0.00014369999999999997,
|
| 193 |
+
"loss": 5.6399181365966795,
|
| 194 |
"step": 480
|
| 195 |
},
|
| 196 |
{
|
| 197 |
+
"epoch": 0.008426019548365353,
|
| 198 |
+
"grad_norm": 1.1796875,
|
| 199 |
+
"learning_rate": 0.00014969999999999998,
|
| 200 |
+
"loss": 5.560699081420898,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 201 |
"step": 500
|
| 202 |
},
|
| 203 |
{
|
| 204 |
+
"epoch": 0.008763060330299966,
|
| 205 |
+
"grad_norm": 2.328125,
|
| 206 |
+
"learning_rate": 0.0001557,
|
| 207 |
+
"loss": 5.474863433837891,
|
| 208 |
"step": 520
|
| 209 |
},
|
| 210 |
{
|
| 211 |
+
"epoch": 0.00910010111223458,
|
| 212 |
+
"grad_norm": 1.125,
|
| 213 |
+
"learning_rate": 0.0001617,
|
| 214 |
+
"loss": 5.396588516235352,
|
| 215 |
"step": 540
|
| 216 |
},
|
| 217 |
{
|
| 218 |
+
"epoch": 0.009437141894169195,
|
| 219 |
+
"grad_norm": 1.6484375,
|
| 220 |
+
"learning_rate": 0.0001677,
|
| 221 |
+
"loss": 5.331023406982422,
|
| 222 |
"step": 560
|
| 223 |
},
|
| 224 |
{
|
| 225 |
+
"epoch": 0.00977418267610381,
|
| 226 |
+
"grad_norm": 1.0703125,
|
| 227 |
+
"learning_rate": 0.00017369999999999997,
|
| 228 |
+
"loss": 5.257175445556641,
|
| 229 |
"step": 580
|
| 230 |
},
|
| 231 |
{
|
| 232 |
+
"epoch": 0.010111223458038422,
|
| 233 |
+
"grad_norm": 2.359375,
|
| 234 |
+
"learning_rate": 0.00017969999999999998,
|
| 235 |
+
"loss": 5.152382659912109,
|
| 236 |
"step": 600
|
| 237 |
},
|
| 238 |
{
|
| 239 |
+
"epoch": 0.010111223458038422,
|
| 240 |
+
"eval_loss": 5.1175713539123535,
|
| 241 |
+
"eval_runtime": 7.457,
|
| 242 |
+
"eval_samples_per_second": 1277.585,
|
| 243 |
+
"eval_steps_per_second": 0.939,
|
| 244 |
"step": 600
|
| 245 |
}
|
| 246 |
],
|
| 247 |
"logging_steps": 20,
|
| 248 |
+
"max_steps": 5000,
|
| 249 |
"num_input_tokens_seen": 0,
|
| 250 |
"num_train_epochs": 1,
|
| 251 |
"save_steps": 100,
|
|
|
|
| 261 |
"attributes": {}
|
| 262 |
}
|
| 263 |
},
|
| 264 |
+
"total_flos": 14521938739200.0,
|
| 265 |
+
"train_batch_size": 16,
|
| 266 |
"trial_name": null,
|
| 267 |
"trial_params": null
|
| 268 |
}
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-600/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-700/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 2036216
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d88313a11f98633534da3abe872a0a8c638d686b7802d4bc680c383765235f92
|
| 3 |
size 2036216
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-700/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4089360
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:171503e70341506b7b06a73f6ad11a065570acb4569cdd00504cb10a5893085b
|
| 3 |
size 4089360
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-700/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14244
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f37c40ce327861a7ca13b719d3aa37510a143368b6e74358bdb14becb3899e1e
|
| 3 |
size 14244
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-700/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:00ed284532aedfadb0830a4afe3ecb31332daef31ec310763e778ba54396ad39
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-700/trainer_state.json
CHANGED
|
@@ -2,317 +2,285 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
-
"eval_steps":
|
| 7 |
"global_step": 700,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
-
"epoch": 0.
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
-
"epoch": 0.
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate":
|
| 23 |
-
"loss": 8.
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"epoch": 0.
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate":
|
| 30 |
-
"loss": 8.
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
-
"epoch": 0.
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate":
|
| 37 |
-
"loss":
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
-
"epoch": 0.
|
| 42 |
-
"grad_norm": 1.
|
| 43 |
-
"learning_rate":
|
| 44 |
-
"loss":
|
| 45 |
-
"step": 100
|
| 46 |
-
},
|
| 47 |
-
{
|
| 48 |
-
"epoch": 0.003370407819346141,
|
| 49 |
-
"eval_loss": 7.687252998352051,
|
| 50 |
-
"eval_runtime": 7.4625,
|
| 51 |
-
"eval_samples_per_second": 1276.648,
|
| 52 |
-
"eval_steps_per_second": 0.938,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
-
"epoch": 0.
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate":
|
| 59 |
-
"loss": 7.
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
-
"epoch": 0.
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate":
|
| 66 |
-
"loss": 7.
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
-
"epoch": 0.
|
| 71 |
-
"grad_norm": 1.
|
| 72 |
-
"learning_rate":
|
| 73 |
-
"loss": 7.
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
-
"epoch": 0.
|
| 78 |
-
"grad_norm": 1.
|
| 79 |
-
"learning_rate":
|
| 80 |
-
"loss":
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
-
"epoch": 0.
|
| 85 |
-
"grad_norm": 1.
|
| 86 |
-
"learning_rate":
|
| 87 |
-
"loss":
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
-
"epoch": 0.
|
| 92 |
-
"eval_loss":
|
| 93 |
-
"eval_runtime": 7.
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
-
"epoch": 0.
|
| 100 |
-
"grad_norm": 1.
|
| 101 |
-
"learning_rate":
|
| 102 |
-
"loss":
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
-
"epoch": 0.
|
| 107 |
-
"grad_norm":
|
| 108 |
-
"learning_rate":
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
-
"epoch": 0.
|
| 114 |
-
"grad_norm":
|
| 115 |
-
"learning_rate":
|
| 116 |
-
"loss": 6.
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
-
"epoch": 0.
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate":
|
| 123 |
-
"loss":
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
-
"epoch": 0.
|
| 128 |
-
"grad_norm": 1.
|
| 129 |
-
"learning_rate":
|
| 130 |
-
"loss":
|
| 131 |
-
"step": 300
|
| 132 |
-
},
|
| 133 |
-
{
|
| 134 |
-
"epoch": 0.010111223458038422,
|
| 135 |
-
"eval_loss": 5.60944938659668,
|
| 136 |
-
"eval_runtime": 7.4311,
|
| 137 |
-
"eval_samples_per_second": 1282.036,
|
| 138 |
-
"eval_steps_per_second": 0.942,
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
-
"epoch": 0.
|
| 143 |
-
"grad_norm":
|
| 144 |
-
"learning_rate":
|
| 145 |
-
"loss":
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
-
"epoch": 0.
|
| 150 |
-
"grad_norm": 1.
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss":
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
-
"epoch": 0.
|
| 157 |
-
"grad_norm": 1.
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss":
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
-
"epoch": 0.
|
| 164 |
-
"grad_norm": 1.
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss":
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
-
"epoch": 0.
|
| 171 |
-
"grad_norm": 1.
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss":
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
-
"epoch": 0.
|
| 178 |
-
"eval_loss":
|
| 179 |
-
"eval_runtime": 7.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
-
"epoch": 0.
|
| 186 |
-
"grad_norm": 0.
|
| 187 |
-
"learning_rate": 0.
|
| 188 |
-
"loss":
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
-
"epoch": 0.
|
| 193 |
-
"grad_norm": 1.
|
| 194 |
-
"learning_rate": 0.
|
| 195 |
-
"loss":
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
-
"epoch": 0.
|
| 200 |
-
"grad_norm":
|
| 201 |
-
"learning_rate": 0.
|
| 202 |
-
"loss":
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
-
"epoch": 0.
|
| 207 |
-
"grad_norm":
|
| 208 |
-
"learning_rate": 0.
|
| 209 |
-
"loss":
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
-
"epoch": 0.
|
| 214 |
-
"grad_norm": 1.
|
| 215 |
-
"learning_rate": 0.
|
| 216 |
-
"loss":
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
-
"epoch": 0.
|
| 221 |
-
"
|
| 222 |
-
"
|
| 223 |
-
"
|
| 224 |
-
"eval_steps_per_second": 0.947,
|
| 225 |
-
"step": 500
|
| 226 |
-
},
|
| 227 |
-
{
|
| 228 |
-
"epoch": 0.01752612066059993,
|
| 229 |
-
"grad_norm": 1.359375,
|
| 230 |
-
"learning_rate": 0.0003,
|
| 231 |
-
"loss": 4.573441696166992,
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
-
"epoch": 0.
|
| 236 |
-
"grad_norm": 1.
|
| 237 |
-
"learning_rate": 0.
|
| 238 |
-
"loss":
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
-
"epoch": 0.
|
| 243 |
-
"grad_norm": 1.
|
| 244 |
-
"learning_rate": 0.
|
| 245 |
-
"loss":
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
-
"epoch": 0.
|
| 250 |
-
"grad_norm": 1.
|
| 251 |
-
"learning_rate": 0.
|
| 252 |
-
"loss":
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
-
"epoch": 0.
|
| 257 |
-
"grad_norm":
|
| 258 |
-
"learning_rate": 0.
|
| 259 |
-
"loss":
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
-
"epoch": 0.
|
| 264 |
-
"eval_loss":
|
| 265 |
-
"eval_runtime": 7.
|
| 266 |
-
"eval_samples_per_second":
|
| 267 |
-
"eval_steps_per_second": 0.
|
| 268 |
"step": 600
|
| 269 |
},
|
| 270 |
{
|
| 271 |
-
"epoch": 0.
|
| 272 |
-
"grad_norm": 1.
|
| 273 |
-
"learning_rate": 0.
|
| 274 |
-
"loss":
|
| 275 |
"step": 620
|
| 276 |
},
|
| 277 |
{
|
| 278 |
-
"epoch": 0.
|
| 279 |
-
"grad_norm": 1.
|
| 280 |
-
"learning_rate": 0.
|
| 281 |
-
"loss":
|
| 282 |
"step": 640
|
| 283 |
},
|
| 284 |
{
|
| 285 |
-
"epoch": 0.
|
| 286 |
-
"grad_norm":
|
| 287 |
-
"learning_rate": 0.
|
| 288 |
-
"loss": 4.
|
| 289 |
"step": 660
|
| 290 |
},
|
| 291 |
{
|
| 292 |
-
"epoch": 0.
|
| 293 |
-
"grad_norm": 2.
|
| 294 |
-
"learning_rate": 0.
|
| 295 |
-
"loss": 4.
|
| 296 |
"step": 680
|
| 297 |
},
|
| 298 |
{
|
| 299 |
-
"epoch": 0.
|
| 300 |
-
"grad_norm": 1.
|
| 301 |
-
"learning_rate": 0.
|
| 302 |
-
"loss": 4.
|
| 303 |
-
"step": 700
|
| 304 |
-
},
|
| 305 |
-
{
|
| 306 |
-
"epoch": 0.023592854735422986,
|
| 307 |
-
"eval_loss": 4.144440650939941,
|
| 308 |
-
"eval_runtime": 7.5426,
|
| 309 |
-
"eval_samples_per_second": 1263.095,
|
| 310 |
-
"eval_steps_per_second": 0.928,
|
| 311 |
"step": 700
|
| 312 |
}
|
| 313 |
],
|
| 314 |
"logging_steps": 20,
|
| 315 |
-
"max_steps":
|
| 316 |
"num_input_tokens_seen": 0,
|
| 317 |
"num_train_epochs": 1,
|
| 318 |
"save_steps": 100,
|
|
@@ -328,8 +296,8 @@
|
|
| 328 |
"attributes": {}
|
| 329 |
}
|
| 330 |
},
|
| 331 |
-
"total_flos":
|
| 332 |
-
"train_batch_size":
|
| 333 |
"trial_name": null,
|
| 334 |
"trial_params": null
|
| 335 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.011796427367711493,
|
| 6 |
+
"eval_steps": 200,
|
| 7 |
"global_step": 700,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
+
"epoch": 0.0003370407819346141,
|
| 14 |
+
"grad_norm": 1.4140625,
|
| 15 |
+
"learning_rate": 5.7e-06,
|
| 16 |
+
"loss": 8.324227142333985,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
+
"epoch": 0.0006740815638692282,
|
| 21 |
+
"grad_norm": 1.5390625,
|
| 22 |
+
"learning_rate": 1.17e-05,
|
| 23 |
+
"loss": 8.318060302734375,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
+
"epoch": 0.0010111223458038423,
|
| 28 |
+
"grad_norm": 1.609375,
|
| 29 |
+
"learning_rate": 1.7699999999999997e-05,
|
| 30 |
+
"loss": 8.293100738525391,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
+
"epoch": 0.0013481631277384564,
|
| 35 |
+
"grad_norm": 1.7578125,
|
| 36 |
+
"learning_rate": 2.3699999999999997e-05,
|
| 37 |
+
"loss": 8.227334594726562,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
+
"epoch": 0.0016852039096730705,
|
| 42 |
+
"grad_norm": 1.5234375,
|
| 43 |
+
"learning_rate": 2.97e-05,
|
| 44 |
+
"loss": 8.111893463134766,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
+
"epoch": 0.0020222446916076846,
|
| 49 |
+
"grad_norm": 1.3046875,
|
| 50 |
+
"learning_rate": 3.5699999999999994e-05,
|
| 51 |
+
"loss": 7.976696014404297,
|
| 52 |
"step": 120
|
| 53 |
},
|
| 54 |
{
|
| 55 |
+
"epoch": 0.0023592854735422987,
|
| 56 |
+
"grad_norm": 1.296875,
|
| 57 |
+
"learning_rate": 4.17e-05,
|
| 58 |
+
"loss": 7.851339721679688,
|
| 59 |
"step": 140
|
| 60 |
},
|
| 61 |
{
|
| 62 |
+
"epoch": 0.002696326255476913,
|
| 63 |
+
"grad_norm": 1.3046875,
|
| 64 |
+
"learning_rate": 4.7699999999999994e-05,
|
| 65 |
+
"loss": 7.724923706054687,
|
| 66 |
"step": 160
|
| 67 |
},
|
| 68 |
{
|
| 69 |
+
"epoch": 0.003033367037411527,
|
| 70 |
+
"grad_norm": 1.3125,
|
| 71 |
+
"learning_rate": 5.369999999999999e-05,
|
| 72 |
+
"loss": 7.58428726196289,
|
| 73 |
"step": 180
|
| 74 |
},
|
| 75 |
{
|
| 76 |
+
"epoch": 0.003370407819346141,
|
| 77 |
+
"grad_norm": 1.25,
|
| 78 |
+
"learning_rate": 5.97e-05,
|
| 79 |
+
"loss": 7.439914703369141,
|
| 80 |
"step": 200
|
| 81 |
},
|
| 82 |
{
|
| 83 |
+
"epoch": 0.003370407819346141,
|
| 84 |
+
"eval_loss": 7.356490135192871,
|
| 85 |
+
"eval_runtime": 7.5119,
|
| 86 |
+
"eval_samples_per_second": 1268.257,
|
| 87 |
+
"eval_steps_per_second": 0.932,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
+
"epoch": 0.003707448601280755,
|
| 92 |
+
"grad_norm": 1.25,
|
| 93 |
+
"learning_rate": 6.57e-05,
|
| 94 |
+
"loss": 7.283377075195313,
|
| 95 |
"step": 220
|
| 96 |
},
|
| 97 |
{
|
| 98 |
+
"epoch": 0.004044489383215369,
|
| 99 |
+
"grad_norm": 1.25,
|
| 100 |
+
"learning_rate": 7.17e-05,
|
| 101 |
+
"loss": 7.127851104736328,
|
| 102 |
"step": 240
|
| 103 |
},
|
| 104 |
{
|
| 105 |
+
"epoch": 0.004381530165149983,
|
| 106 |
+
"grad_norm": 1.2109375,
|
| 107 |
+
"learning_rate": 7.769999999999999e-05,
|
| 108 |
+
"loss": 6.964313507080078,
|
| 109 |
"step": 260
|
| 110 |
},
|
| 111 |
{
|
| 112 |
+
"epoch": 0.0047185709470845974,
|
| 113 |
+
"grad_norm": 1.171875,
|
| 114 |
+
"learning_rate": 8.37e-05,
|
| 115 |
+
"loss": 6.807338714599609,
|
| 116 |
"step": 280
|
| 117 |
},
|
| 118 |
{
|
| 119 |
+
"epoch": 0.005055611729019211,
|
| 120 |
+
"grad_norm": 1.1328125,
|
| 121 |
+
"learning_rate": 8.969999999999998e-05,
|
| 122 |
+
"loss": 6.667655181884766,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 123 |
"step": 300
|
| 124 |
},
|
| 125 |
{
|
| 126 |
+
"epoch": 0.005392652510953826,
|
| 127 |
+
"grad_norm": 1.109375,
|
| 128 |
+
"learning_rate": 9.57e-05,
|
| 129 |
+
"loss": 6.523377227783203,
|
| 130 |
"step": 320
|
| 131 |
},
|
| 132 |
{
|
| 133 |
+
"epoch": 0.005729693292888439,
|
| 134 |
+
"grad_norm": 1.1171875,
|
| 135 |
+
"learning_rate": 0.00010169999999999999,
|
| 136 |
+
"loss": 6.383005142211914,
|
| 137 |
"step": 340
|
| 138 |
},
|
| 139 |
{
|
| 140 |
+
"epoch": 0.006066734074823054,
|
| 141 |
+
"grad_norm": 1.6875,
|
| 142 |
+
"learning_rate": 0.00010769999999999999,
|
| 143 |
+
"loss": 6.261091232299805,
|
| 144 |
"step": 360
|
| 145 |
},
|
| 146 |
{
|
| 147 |
+
"epoch": 0.0064037748567576675,
|
| 148 |
+
"grad_norm": 1.140625,
|
| 149 |
+
"learning_rate": 0.00011369999999999999,
|
| 150 |
+
"loss": 6.122833251953125,
|
| 151 |
"step": 380
|
| 152 |
},
|
| 153 |
{
|
| 154 |
+
"epoch": 0.006740815638692282,
|
| 155 |
+
"grad_norm": 1.3984375,
|
| 156 |
+
"learning_rate": 0.0001197,
|
| 157 |
+
"loss": 6.019657897949219,
|
| 158 |
"step": 400
|
| 159 |
},
|
| 160 |
{
|
| 161 |
+
"epoch": 0.006740815638692282,
|
| 162 |
+
"eval_loss": 5.966014385223389,
|
| 163 |
+
"eval_runtime": 7.516,
|
| 164 |
+
"eval_samples_per_second": 1267.556,
|
| 165 |
+
"eval_steps_per_second": 0.931,
|
| 166 |
"step": 400
|
| 167 |
},
|
| 168 |
{
|
| 169 |
+
"epoch": 0.007077856420626896,
|
| 170 |
+
"grad_norm": 0.98046875,
|
| 171 |
+
"learning_rate": 0.0001257,
|
| 172 |
+
"loss": 5.9373779296875,
|
| 173 |
"step": 420
|
| 174 |
},
|
| 175 |
{
|
| 176 |
+
"epoch": 0.00741489720256151,
|
| 177 |
+
"grad_norm": 1.6328125,
|
| 178 |
+
"learning_rate": 0.00013169999999999998,
|
| 179 |
+
"loss": 5.839211273193359,
|
| 180 |
"step": 440
|
| 181 |
},
|
| 182 |
{
|
| 183 |
+
"epoch": 0.007751937984496124,
|
| 184 |
+
"grad_norm": 0.9609375,
|
| 185 |
+
"learning_rate": 0.00013769999999999999,
|
| 186 |
+
"loss": 5.740922927856445,
|
| 187 |
"step": 460
|
| 188 |
},
|
| 189 |
{
|
| 190 |
+
"epoch": 0.008088978766430738,
|
| 191 |
+
"grad_norm": 0.90234375,
|
| 192 |
+
"learning_rate": 0.00014369999999999997,
|
| 193 |
+
"loss": 5.6399181365966795,
|
| 194 |
"step": 480
|
| 195 |
},
|
| 196 |
{
|
| 197 |
+
"epoch": 0.008426019548365353,
|
| 198 |
+
"grad_norm": 1.1796875,
|
| 199 |
+
"learning_rate": 0.00014969999999999998,
|
| 200 |
+
"loss": 5.560699081420898,
|
| 201 |
"step": 500
|
| 202 |
},
|
| 203 |
{
|
| 204 |
+
"epoch": 0.008763060330299966,
|
| 205 |
+
"grad_norm": 2.328125,
|
| 206 |
+
"learning_rate": 0.0001557,
|
| 207 |
+
"loss": 5.474863433837891,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 208 |
"step": 520
|
| 209 |
},
|
| 210 |
{
|
| 211 |
+
"epoch": 0.00910010111223458,
|
| 212 |
+
"grad_norm": 1.125,
|
| 213 |
+
"learning_rate": 0.0001617,
|
| 214 |
+
"loss": 5.396588516235352,
|
| 215 |
"step": 540
|
| 216 |
},
|
| 217 |
{
|
| 218 |
+
"epoch": 0.009437141894169195,
|
| 219 |
+
"grad_norm": 1.6484375,
|
| 220 |
+
"learning_rate": 0.0001677,
|
| 221 |
+
"loss": 5.331023406982422,
|
| 222 |
"step": 560
|
| 223 |
},
|
| 224 |
{
|
| 225 |
+
"epoch": 0.00977418267610381,
|
| 226 |
+
"grad_norm": 1.0703125,
|
| 227 |
+
"learning_rate": 0.00017369999999999997,
|
| 228 |
+
"loss": 5.257175445556641,
|
| 229 |
"step": 580
|
| 230 |
},
|
| 231 |
{
|
| 232 |
+
"epoch": 0.010111223458038422,
|
| 233 |
+
"grad_norm": 2.359375,
|
| 234 |
+
"learning_rate": 0.00017969999999999998,
|
| 235 |
+
"loss": 5.152382659912109,
|
| 236 |
"step": 600
|
| 237 |
},
|
| 238 |
{
|
| 239 |
+
"epoch": 0.010111223458038422,
|
| 240 |
+
"eval_loss": 5.1175713539123535,
|
| 241 |
+
"eval_runtime": 7.457,
|
| 242 |
+
"eval_samples_per_second": 1277.585,
|
| 243 |
+
"eval_steps_per_second": 0.939,
|
| 244 |
"step": 600
|
| 245 |
},
|
| 246 |
{
|
| 247 |
+
"epoch": 0.010448264239973037,
|
| 248 |
+
"grad_norm": 1.375,
|
| 249 |
+
"learning_rate": 0.0001857,
|
| 250 |
+
"loss": 5.096985244750977,
|
| 251 |
"step": 620
|
| 252 |
},
|
| 253 |
{
|
| 254 |
+
"epoch": 0.010785305021907651,
|
| 255 |
+
"grad_norm": 1.421875,
|
| 256 |
+
"learning_rate": 0.0001917,
|
| 257 |
+
"loss": 5.019801330566406,
|
| 258 |
"step": 640
|
| 259 |
},
|
| 260 |
{
|
| 261 |
+
"epoch": 0.011122345803842264,
|
| 262 |
+
"grad_norm": 2.125,
|
| 263 |
+
"learning_rate": 0.00019769999999999998,
|
| 264 |
+
"loss": 4.979957962036133,
|
| 265 |
"step": 660
|
| 266 |
},
|
| 267 |
{
|
| 268 |
+
"epoch": 0.011459386585776879,
|
| 269 |
+
"grad_norm": 2.203125,
|
| 270 |
+
"learning_rate": 0.0002037,
|
| 271 |
+
"loss": 4.945652008056641,
|
| 272 |
"step": 680
|
| 273 |
},
|
| 274 |
{
|
| 275 |
+
"epoch": 0.011796427367711493,
|
| 276 |
+
"grad_norm": 1.8515625,
|
| 277 |
+
"learning_rate": 0.00020969999999999997,
|
| 278 |
+
"loss": 4.866051483154297,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 279 |
"step": 700
|
| 280 |
}
|
| 281 |
],
|
| 282 |
"logging_steps": 20,
|
| 283 |
+
"max_steps": 5000,
|
| 284 |
"num_input_tokens_seen": 0,
|
| 285 |
"num_train_epochs": 1,
|
| 286 |
"save_steps": 100,
|
|
|
|
| 296 |
"attributes": {}
|
| 297 |
}
|
| 298 |
},
|
| 299 |
+
"total_flos": 16942261862400.0,
|
| 300 |
+
"train_batch_size": 16,
|
| 301 |
"trial_name": null,
|
| 302 |
"trial_params": null
|
| 303 |
}
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-700/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-800/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 2036216
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0edb3ed95464c61de67c9cffe7938ede900e73d8fafa0983797e3d5ec92328f7
|
| 3 |
size 2036216
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-800/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4089360
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:522169e9d2be0062e3eda5870ee10162fec7ef76f98244b4288d242b0044761a
|
| 3 |
size 4089360
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-800/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14244
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ecefbb3f17bb76b6655eb0157c98b5287c17fa4b4c72a6b9068b0823ce9fd18d
|
| 3 |
size 14244
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-800/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c39c131425d66759f6276fee23b3a54e5b3da37f6f8ee3949f69449c73bf15ee
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-800/trainer_state.json
CHANGED
|
@@ -2,360 +2,328 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
-
"eval_steps":
|
| 7 |
"global_step": 800,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
-
"epoch": 0.
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
-
"epoch": 0.
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate":
|
| 23 |
-
"loss": 8.
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"epoch": 0.
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate":
|
| 30 |
-
"loss": 8.
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
-
"epoch": 0.
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate":
|
| 37 |
-
"loss":
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
-
"epoch": 0.
|
| 42 |
-
"grad_norm": 1.
|
| 43 |
-
"learning_rate":
|
| 44 |
-
"loss":
|
| 45 |
-
"step": 100
|
| 46 |
-
},
|
| 47 |
-
{
|
| 48 |
-
"epoch": 0.003370407819346141,
|
| 49 |
-
"eval_loss": 7.687252998352051,
|
| 50 |
-
"eval_runtime": 7.4625,
|
| 51 |
-
"eval_samples_per_second": 1276.648,
|
| 52 |
-
"eval_steps_per_second": 0.938,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
-
"epoch": 0.
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate":
|
| 59 |
-
"loss": 7.
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
-
"epoch": 0.
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate":
|
| 66 |
-
"loss": 7.
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
-
"epoch": 0.
|
| 71 |
-
"grad_norm": 1.
|
| 72 |
-
"learning_rate":
|
| 73 |
-
"loss": 7.
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
-
"epoch": 0.
|
| 78 |
-
"grad_norm": 1.
|
| 79 |
-
"learning_rate":
|
| 80 |
-
"loss":
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
-
"epoch": 0.
|
| 85 |
-
"grad_norm": 1.
|
| 86 |
-
"learning_rate":
|
| 87 |
-
"loss":
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
-
"epoch": 0.
|
| 92 |
-
"eval_loss":
|
| 93 |
-
"eval_runtime": 7.
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
-
"epoch": 0.
|
| 100 |
-
"grad_norm": 1.
|
| 101 |
-
"learning_rate":
|
| 102 |
-
"loss":
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
-
"epoch": 0.
|
| 107 |
-
"grad_norm":
|
| 108 |
-
"learning_rate":
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
-
"epoch": 0.
|
| 114 |
-
"grad_norm":
|
| 115 |
-
"learning_rate":
|
| 116 |
-
"loss": 6.
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
-
"epoch": 0.
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate":
|
| 123 |
-
"loss":
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
-
"epoch": 0.
|
| 128 |
-
"grad_norm": 1.
|
| 129 |
-
"learning_rate":
|
| 130 |
-
"loss":
|
| 131 |
-
"step": 300
|
| 132 |
-
},
|
| 133 |
-
{
|
| 134 |
-
"epoch": 0.010111223458038422,
|
| 135 |
-
"eval_loss": 5.60944938659668,
|
| 136 |
-
"eval_runtime": 7.4311,
|
| 137 |
-
"eval_samples_per_second": 1282.036,
|
| 138 |
-
"eval_steps_per_second": 0.942,
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
-
"epoch": 0.
|
| 143 |
-
"grad_norm":
|
| 144 |
-
"learning_rate":
|
| 145 |
-
"loss":
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
-
"epoch": 0.
|
| 150 |
-
"grad_norm": 1.
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss":
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
-
"epoch": 0.
|
| 157 |
-
"grad_norm": 1.
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss":
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
-
"epoch": 0.
|
| 164 |
-
"grad_norm": 1.
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss":
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
-
"epoch": 0.
|
| 171 |
-
"grad_norm": 1.
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss":
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
-
"epoch": 0.
|
| 178 |
-
"eval_loss":
|
| 179 |
-
"eval_runtime": 7.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
-
"epoch": 0.
|
| 186 |
-
"grad_norm": 0.
|
| 187 |
-
"learning_rate": 0.
|
| 188 |
-
"loss":
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
-
"epoch": 0.
|
| 193 |
-
"grad_norm": 1.
|
| 194 |
-
"learning_rate": 0.
|
| 195 |
-
"loss":
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
-
"epoch": 0.
|
| 200 |
-
"grad_norm":
|
| 201 |
-
"learning_rate": 0.
|
| 202 |
-
"loss":
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
-
"epoch": 0.
|
| 207 |
-
"grad_norm":
|
| 208 |
-
"learning_rate": 0.
|
| 209 |
-
"loss":
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
-
"epoch": 0.
|
| 214 |
-
"grad_norm": 1.
|
| 215 |
-
"learning_rate": 0.
|
| 216 |
-
"loss":
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
-
"epoch": 0.
|
| 221 |
-
"
|
| 222 |
-
"
|
| 223 |
-
"
|
| 224 |
-
"eval_steps_per_second": 0.947,
|
| 225 |
-
"step": 500
|
| 226 |
-
},
|
| 227 |
-
{
|
| 228 |
-
"epoch": 0.01752612066059993,
|
| 229 |
-
"grad_norm": 1.359375,
|
| 230 |
-
"learning_rate": 0.0003,
|
| 231 |
-
"loss": 4.573441696166992,
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
-
"epoch": 0.
|
| 236 |
-
"grad_norm": 1.
|
| 237 |
-
"learning_rate": 0.
|
| 238 |
-
"loss":
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
-
"epoch": 0.
|
| 243 |
-
"grad_norm": 1.
|
| 244 |
-
"learning_rate": 0.
|
| 245 |
-
"loss":
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
-
"epoch": 0.
|
| 250 |
-
"grad_norm": 1.
|
| 251 |
-
"learning_rate": 0.
|
| 252 |
-
"loss":
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
-
"epoch": 0.
|
| 257 |
-
"grad_norm":
|
| 258 |
-
"learning_rate": 0.
|
| 259 |
-
"loss":
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
-
"epoch": 0.
|
| 264 |
-
"eval_loss":
|
| 265 |
-
"eval_runtime": 7.
|
| 266 |
-
"eval_samples_per_second":
|
| 267 |
-
"eval_steps_per_second": 0.
|
| 268 |
"step": 600
|
| 269 |
},
|
| 270 |
{
|
| 271 |
-
"epoch": 0.
|
| 272 |
-
"grad_norm": 1.
|
| 273 |
-
"learning_rate": 0.
|
| 274 |
-
"loss":
|
| 275 |
"step": 620
|
| 276 |
},
|
| 277 |
{
|
| 278 |
-
"epoch": 0.
|
| 279 |
-
"grad_norm": 1.
|
| 280 |
-
"learning_rate": 0.
|
| 281 |
-
"loss":
|
| 282 |
"step": 640
|
| 283 |
},
|
| 284 |
{
|
| 285 |
-
"epoch": 0.
|
| 286 |
-
"grad_norm":
|
| 287 |
-
"learning_rate": 0.
|
| 288 |
-
"loss": 4.
|
| 289 |
"step": 660
|
| 290 |
},
|
| 291 |
{
|
| 292 |
-
"epoch": 0.
|
| 293 |
-
"grad_norm": 2.
|
| 294 |
-
"learning_rate": 0.
|
| 295 |
-
"loss": 4.
|
| 296 |
"step": 680
|
| 297 |
},
|
| 298 |
{
|
| 299 |
-
"epoch": 0.
|
| 300 |
-
"grad_norm": 1.
|
| 301 |
-
"learning_rate": 0.
|
| 302 |
-
"loss": 4.
|
| 303 |
"step": 700
|
| 304 |
},
|
| 305 |
{
|
| 306 |
-
"epoch": 0.
|
| 307 |
-
"
|
| 308 |
-
"
|
| 309 |
-
"
|
| 310 |
-
"eval_steps_per_second": 0.928,
|
| 311 |
-
"step": 700
|
| 312 |
-
},
|
| 313 |
-
{
|
| 314 |
-
"epoch": 0.024266936299292215,
|
| 315 |
-
"grad_norm": 0.98828125,
|
| 316 |
-
"learning_rate": 0.0003,
|
| 317 |
-
"loss": 4.132168579101562,
|
| 318 |
"step": 720
|
| 319 |
},
|
| 320 |
{
|
| 321 |
-
"epoch": 0.
|
| 322 |
-
"grad_norm":
|
| 323 |
-
"learning_rate": 0.
|
| 324 |
-
"loss": 4.
|
| 325 |
"step": 740
|
| 326 |
},
|
| 327 |
{
|
| 328 |
-
"epoch": 0.
|
| 329 |
-
"grad_norm": 2.
|
| 330 |
-
"learning_rate": 0.
|
| 331 |
-
"loss": 4.
|
| 332 |
"step": 760
|
| 333 |
},
|
| 334 |
{
|
| 335 |
-
"epoch": 0.
|
| 336 |
-
"grad_norm": 1.
|
| 337 |
-
"learning_rate": 0.
|
| 338 |
-
"loss": 4.
|
| 339 |
"step": 780
|
| 340 |
},
|
| 341 |
{
|
| 342 |
-
"epoch": 0.
|
| 343 |
-
"grad_norm": 1.
|
| 344 |
-
"learning_rate": 0.
|
| 345 |
-
"loss": 4.
|
| 346 |
"step": 800
|
| 347 |
},
|
| 348 |
{
|
| 349 |
-
"epoch": 0.
|
| 350 |
-
"eval_loss": 4.
|
| 351 |
-
"eval_runtime": 7.
|
| 352 |
-
"eval_samples_per_second":
|
| 353 |
-
"eval_steps_per_second": 0.
|
| 354 |
"step": 800
|
| 355 |
}
|
| 356 |
],
|
| 357 |
"logging_steps": 20,
|
| 358 |
-
"max_steps":
|
| 359 |
"num_input_tokens_seen": 0,
|
| 360 |
"num_train_epochs": 1,
|
| 361 |
"save_steps": 100,
|
|
@@ -371,8 +339,8 @@
|
|
| 371 |
"attributes": {}
|
| 372 |
}
|
| 373 |
},
|
| 374 |
-
"total_flos":
|
| 375 |
-
"train_batch_size":
|
| 376 |
"trial_name": null,
|
| 377 |
"trial_params": null
|
| 378 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.013481631277384564,
|
| 6 |
+
"eval_steps": 200,
|
| 7 |
"global_step": 800,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
+
"epoch": 0.0003370407819346141,
|
| 14 |
+
"grad_norm": 1.4140625,
|
| 15 |
+
"learning_rate": 5.7e-06,
|
| 16 |
+
"loss": 8.324227142333985,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
+
"epoch": 0.0006740815638692282,
|
| 21 |
+
"grad_norm": 1.5390625,
|
| 22 |
+
"learning_rate": 1.17e-05,
|
| 23 |
+
"loss": 8.318060302734375,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
+
"epoch": 0.0010111223458038423,
|
| 28 |
+
"grad_norm": 1.609375,
|
| 29 |
+
"learning_rate": 1.7699999999999997e-05,
|
| 30 |
+
"loss": 8.293100738525391,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
+
"epoch": 0.0013481631277384564,
|
| 35 |
+
"grad_norm": 1.7578125,
|
| 36 |
+
"learning_rate": 2.3699999999999997e-05,
|
| 37 |
+
"loss": 8.227334594726562,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
+
"epoch": 0.0016852039096730705,
|
| 42 |
+
"grad_norm": 1.5234375,
|
| 43 |
+
"learning_rate": 2.97e-05,
|
| 44 |
+
"loss": 8.111893463134766,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
+
"epoch": 0.0020222446916076846,
|
| 49 |
+
"grad_norm": 1.3046875,
|
| 50 |
+
"learning_rate": 3.5699999999999994e-05,
|
| 51 |
+
"loss": 7.976696014404297,
|
| 52 |
"step": 120
|
| 53 |
},
|
| 54 |
{
|
| 55 |
+
"epoch": 0.0023592854735422987,
|
| 56 |
+
"grad_norm": 1.296875,
|
| 57 |
+
"learning_rate": 4.17e-05,
|
| 58 |
+
"loss": 7.851339721679688,
|
| 59 |
"step": 140
|
| 60 |
},
|
| 61 |
{
|
| 62 |
+
"epoch": 0.002696326255476913,
|
| 63 |
+
"grad_norm": 1.3046875,
|
| 64 |
+
"learning_rate": 4.7699999999999994e-05,
|
| 65 |
+
"loss": 7.724923706054687,
|
| 66 |
"step": 160
|
| 67 |
},
|
| 68 |
{
|
| 69 |
+
"epoch": 0.003033367037411527,
|
| 70 |
+
"grad_norm": 1.3125,
|
| 71 |
+
"learning_rate": 5.369999999999999e-05,
|
| 72 |
+
"loss": 7.58428726196289,
|
| 73 |
"step": 180
|
| 74 |
},
|
| 75 |
{
|
| 76 |
+
"epoch": 0.003370407819346141,
|
| 77 |
+
"grad_norm": 1.25,
|
| 78 |
+
"learning_rate": 5.97e-05,
|
| 79 |
+
"loss": 7.439914703369141,
|
| 80 |
"step": 200
|
| 81 |
},
|
| 82 |
{
|
| 83 |
+
"epoch": 0.003370407819346141,
|
| 84 |
+
"eval_loss": 7.356490135192871,
|
| 85 |
+
"eval_runtime": 7.5119,
|
| 86 |
+
"eval_samples_per_second": 1268.257,
|
| 87 |
+
"eval_steps_per_second": 0.932,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
+
"epoch": 0.003707448601280755,
|
| 92 |
+
"grad_norm": 1.25,
|
| 93 |
+
"learning_rate": 6.57e-05,
|
| 94 |
+
"loss": 7.283377075195313,
|
| 95 |
"step": 220
|
| 96 |
},
|
| 97 |
{
|
| 98 |
+
"epoch": 0.004044489383215369,
|
| 99 |
+
"grad_norm": 1.25,
|
| 100 |
+
"learning_rate": 7.17e-05,
|
| 101 |
+
"loss": 7.127851104736328,
|
| 102 |
"step": 240
|
| 103 |
},
|
| 104 |
{
|
| 105 |
+
"epoch": 0.004381530165149983,
|
| 106 |
+
"grad_norm": 1.2109375,
|
| 107 |
+
"learning_rate": 7.769999999999999e-05,
|
| 108 |
+
"loss": 6.964313507080078,
|
| 109 |
"step": 260
|
| 110 |
},
|
| 111 |
{
|
| 112 |
+
"epoch": 0.0047185709470845974,
|
| 113 |
+
"grad_norm": 1.171875,
|
| 114 |
+
"learning_rate": 8.37e-05,
|
| 115 |
+
"loss": 6.807338714599609,
|
| 116 |
"step": 280
|
| 117 |
},
|
| 118 |
{
|
| 119 |
+
"epoch": 0.005055611729019211,
|
| 120 |
+
"grad_norm": 1.1328125,
|
| 121 |
+
"learning_rate": 8.969999999999998e-05,
|
| 122 |
+
"loss": 6.667655181884766,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 123 |
"step": 300
|
| 124 |
},
|
| 125 |
{
|
| 126 |
+
"epoch": 0.005392652510953826,
|
| 127 |
+
"grad_norm": 1.109375,
|
| 128 |
+
"learning_rate": 9.57e-05,
|
| 129 |
+
"loss": 6.523377227783203,
|
| 130 |
"step": 320
|
| 131 |
},
|
| 132 |
{
|
| 133 |
+
"epoch": 0.005729693292888439,
|
| 134 |
+
"grad_norm": 1.1171875,
|
| 135 |
+
"learning_rate": 0.00010169999999999999,
|
| 136 |
+
"loss": 6.383005142211914,
|
| 137 |
"step": 340
|
| 138 |
},
|
| 139 |
{
|
| 140 |
+
"epoch": 0.006066734074823054,
|
| 141 |
+
"grad_norm": 1.6875,
|
| 142 |
+
"learning_rate": 0.00010769999999999999,
|
| 143 |
+
"loss": 6.261091232299805,
|
| 144 |
"step": 360
|
| 145 |
},
|
| 146 |
{
|
| 147 |
+
"epoch": 0.0064037748567576675,
|
| 148 |
+
"grad_norm": 1.140625,
|
| 149 |
+
"learning_rate": 0.00011369999999999999,
|
| 150 |
+
"loss": 6.122833251953125,
|
| 151 |
"step": 380
|
| 152 |
},
|
| 153 |
{
|
| 154 |
+
"epoch": 0.006740815638692282,
|
| 155 |
+
"grad_norm": 1.3984375,
|
| 156 |
+
"learning_rate": 0.0001197,
|
| 157 |
+
"loss": 6.019657897949219,
|
| 158 |
"step": 400
|
| 159 |
},
|
| 160 |
{
|
| 161 |
+
"epoch": 0.006740815638692282,
|
| 162 |
+
"eval_loss": 5.966014385223389,
|
| 163 |
+
"eval_runtime": 7.516,
|
| 164 |
+
"eval_samples_per_second": 1267.556,
|
| 165 |
+
"eval_steps_per_second": 0.931,
|
| 166 |
"step": 400
|
| 167 |
},
|
| 168 |
{
|
| 169 |
+
"epoch": 0.007077856420626896,
|
| 170 |
+
"grad_norm": 0.98046875,
|
| 171 |
+
"learning_rate": 0.0001257,
|
| 172 |
+
"loss": 5.9373779296875,
|
| 173 |
"step": 420
|
| 174 |
},
|
| 175 |
{
|
| 176 |
+
"epoch": 0.00741489720256151,
|
| 177 |
+
"grad_norm": 1.6328125,
|
| 178 |
+
"learning_rate": 0.00013169999999999998,
|
| 179 |
+
"loss": 5.839211273193359,
|
| 180 |
"step": 440
|
| 181 |
},
|
| 182 |
{
|
| 183 |
+
"epoch": 0.007751937984496124,
|
| 184 |
+
"grad_norm": 0.9609375,
|
| 185 |
+
"learning_rate": 0.00013769999999999999,
|
| 186 |
+
"loss": 5.740922927856445,
|
| 187 |
"step": 460
|
| 188 |
},
|
| 189 |
{
|
| 190 |
+
"epoch": 0.008088978766430738,
|
| 191 |
+
"grad_norm": 0.90234375,
|
| 192 |
+
"learning_rate": 0.00014369999999999997,
|
| 193 |
+
"loss": 5.6399181365966795,
|
| 194 |
"step": 480
|
| 195 |
},
|
| 196 |
{
|
| 197 |
+
"epoch": 0.008426019548365353,
|
| 198 |
+
"grad_norm": 1.1796875,
|
| 199 |
+
"learning_rate": 0.00014969999999999998,
|
| 200 |
+
"loss": 5.560699081420898,
|
| 201 |
"step": 500
|
| 202 |
},
|
| 203 |
{
|
| 204 |
+
"epoch": 0.008763060330299966,
|
| 205 |
+
"grad_norm": 2.328125,
|
| 206 |
+
"learning_rate": 0.0001557,
|
| 207 |
+
"loss": 5.474863433837891,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 208 |
"step": 520
|
| 209 |
},
|
| 210 |
{
|
| 211 |
+
"epoch": 0.00910010111223458,
|
| 212 |
+
"grad_norm": 1.125,
|
| 213 |
+
"learning_rate": 0.0001617,
|
| 214 |
+
"loss": 5.396588516235352,
|
| 215 |
"step": 540
|
| 216 |
},
|
| 217 |
{
|
| 218 |
+
"epoch": 0.009437141894169195,
|
| 219 |
+
"grad_norm": 1.6484375,
|
| 220 |
+
"learning_rate": 0.0001677,
|
| 221 |
+
"loss": 5.331023406982422,
|
| 222 |
"step": 560
|
| 223 |
},
|
| 224 |
{
|
| 225 |
+
"epoch": 0.00977418267610381,
|
| 226 |
+
"grad_norm": 1.0703125,
|
| 227 |
+
"learning_rate": 0.00017369999999999997,
|
| 228 |
+
"loss": 5.257175445556641,
|
| 229 |
"step": 580
|
| 230 |
},
|
| 231 |
{
|
| 232 |
+
"epoch": 0.010111223458038422,
|
| 233 |
+
"grad_norm": 2.359375,
|
| 234 |
+
"learning_rate": 0.00017969999999999998,
|
| 235 |
+
"loss": 5.152382659912109,
|
| 236 |
"step": 600
|
| 237 |
},
|
| 238 |
{
|
| 239 |
+
"epoch": 0.010111223458038422,
|
| 240 |
+
"eval_loss": 5.1175713539123535,
|
| 241 |
+
"eval_runtime": 7.457,
|
| 242 |
+
"eval_samples_per_second": 1277.585,
|
| 243 |
+
"eval_steps_per_second": 0.939,
|
| 244 |
"step": 600
|
| 245 |
},
|
| 246 |
{
|
| 247 |
+
"epoch": 0.010448264239973037,
|
| 248 |
+
"grad_norm": 1.375,
|
| 249 |
+
"learning_rate": 0.0001857,
|
| 250 |
+
"loss": 5.096985244750977,
|
| 251 |
"step": 620
|
| 252 |
},
|
| 253 |
{
|
| 254 |
+
"epoch": 0.010785305021907651,
|
| 255 |
+
"grad_norm": 1.421875,
|
| 256 |
+
"learning_rate": 0.0001917,
|
| 257 |
+
"loss": 5.019801330566406,
|
| 258 |
"step": 640
|
| 259 |
},
|
| 260 |
{
|
| 261 |
+
"epoch": 0.011122345803842264,
|
| 262 |
+
"grad_norm": 2.125,
|
| 263 |
+
"learning_rate": 0.00019769999999999998,
|
| 264 |
+
"loss": 4.979957962036133,
|
| 265 |
"step": 660
|
| 266 |
},
|
| 267 |
{
|
| 268 |
+
"epoch": 0.011459386585776879,
|
| 269 |
+
"grad_norm": 2.203125,
|
| 270 |
+
"learning_rate": 0.0002037,
|
| 271 |
+
"loss": 4.945652008056641,
|
| 272 |
"step": 680
|
| 273 |
},
|
| 274 |
{
|
| 275 |
+
"epoch": 0.011796427367711493,
|
| 276 |
+
"grad_norm": 1.8515625,
|
| 277 |
+
"learning_rate": 0.00020969999999999997,
|
| 278 |
+
"loss": 4.866051483154297,
|
| 279 |
"step": 700
|
| 280 |
},
|
| 281 |
{
|
| 282 |
+
"epoch": 0.012133468149646108,
|
| 283 |
+
"grad_norm": 1.625,
|
| 284 |
+
"learning_rate": 0.00021569999999999998,
|
| 285 |
+
"loss": 4.841766357421875,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 286 |
"step": 720
|
| 287 |
},
|
| 288 |
{
|
| 289 |
+
"epoch": 0.01247050893158072,
|
| 290 |
+
"grad_norm": 3.40625,
|
| 291 |
+
"learning_rate": 0.00022169999999999997,
|
| 292 |
+
"loss": 4.798672103881836,
|
| 293 |
"step": 740
|
| 294 |
},
|
| 295 |
{
|
| 296 |
+
"epoch": 0.012807549713515335,
|
| 297 |
+
"grad_norm": 2.21875,
|
| 298 |
+
"learning_rate": 0.00022769999999999998,
|
| 299 |
+
"loss": 4.768531036376953,
|
| 300 |
"step": 760
|
| 301 |
},
|
| 302 |
{
|
| 303 |
+
"epoch": 0.01314459049544995,
|
| 304 |
+
"grad_norm": 1.640625,
|
| 305 |
+
"learning_rate": 0.0002337,
|
| 306 |
+
"loss": 4.73585319519043,
|
| 307 |
"step": 780
|
| 308 |
},
|
| 309 |
{
|
| 310 |
+
"epoch": 0.013481631277384564,
|
| 311 |
+
"grad_norm": 1.6875,
|
| 312 |
+
"learning_rate": 0.0002397,
|
| 313 |
+
"loss": 4.697021865844727,
|
| 314 |
"step": 800
|
| 315 |
},
|
| 316 |
{
|
| 317 |
+
"epoch": 0.013481631277384564,
|
| 318 |
+
"eval_loss": 4.676848411560059,
|
| 319 |
+
"eval_runtime": 7.4394,
|
| 320 |
+
"eval_samples_per_second": 1280.62,
|
| 321 |
+
"eval_steps_per_second": 0.941,
|
| 322 |
"step": 800
|
| 323 |
}
|
| 324 |
],
|
| 325 |
"logging_steps": 20,
|
| 326 |
+
"max_steps": 5000,
|
| 327 |
"num_input_tokens_seen": 0,
|
| 328 |
"num_train_epochs": 1,
|
| 329 |
"save_steps": 100,
|
|
|
|
| 339 |
"attributes": {}
|
| 340 |
}
|
| 341 |
},
|
| 342 |
+
"total_flos": 19362584985600.0,
|
| 343 |
+
"train_batch_size": 16,
|
| 344 |
"trial_name": null,
|
| 345 |
"trial_params": null
|
| 346 |
}
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-800/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 2036216
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8f12b34cc7fdfd828fc5af59ab51663686e06ecf81a5c0bcaa560f7f1e344f6f
|
| 3 |
size 2036216
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4089360
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4d775d3b8c1660cfecccbcb10103788b0d29af15ab335fb8688ec578610c2811
|
| 3 |
size 4089360
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14244
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ecefbb3f17bb76b6655eb0157c98b5287c17fa4b4c72a6b9068b0823ce9fd18d
|
| 3 |
size 14244
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5bfb7b9c091ab83b2f167c61a3ae9a0c249f254498c8844e764454e070c9b77a
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/trainer_state.json
CHANGED
|
@@ -2,403 +2,363 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
-
"eval_steps":
|
| 7 |
"global_step": 900,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
-
"epoch": 0.
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
-
"epoch": 0.
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate":
|
| 23 |
-
"loss": 8.
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"epoch": 0.
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate":
|
| 30 |
-
"loss": 8.
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
-
"epoch": 0.
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate":
|
| 37 |
-
"loss":
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
-
"epoch": 0.
|
| 42 |
-
"grad_norm": 1.
|
| 43 |
-
"learning_rate":
|
| 44 |
-
"loss":
|
| 45 |
-
"step": 100
|
| 46 |
-
},
|
| 47 |
-
{
|
| 48 |
-
"epoch": 0.003370407819346141,
|
| 49 |
-
"eval_loss": 7.687252998352051,
|
| 50 |
-
"eval_runtime": 7.4625,
|
| 51 |
-
"eval_samples_per_second": 1276.648,
|
| 52 |
-
"eval_steps_per_second": 0.938,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
-
"epoch": 0.
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate":
|
| 59 |
-
"loss": 7.
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
-
"epoch": 0.
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate":
|
| 66 |
-
"loss": 7.
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
-
"epoch": 0.
|
| 71 |
-
"grad_norm": 1.
|
| 72 |
-
"learning_rate":
|
| 73 |
-
"loss": 7.
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
-
"epoch": 0.
|
| 78 |
-
"grad_norm": 1.
|
| 79 |
-
"learning_rate":
|
| 80 |
-
"loss":
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
-
"epoch": 0.
|
| 85 |
-
"grad_norm": 1.
|
| 86 |
-
"learning_rate":
|
| 87 |
-
"loss":
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
-
"epoch": 0.
|
| 92 |
-
"eval_loss":
|
| 93 |
-
"eval_runtime": 7.
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
-
"epoch": 0.
|
| 100 |
-
"grad_norm": 1.
|
| 101 |
-
"learning_rate":
|
| 102 |
-
"loss":
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
-
"epoch": 0.
|
| 107 |
-
"grad_norm":
|
| 108 |
-
"learning_rate":
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
-
"epoch": 0.
|
| 114 |
-
"grad_norm":
|
| 115 |
-
"learning_rate":
|
| 116 |
-
"loss": 6.
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
-
"epoch": 0.
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate":
|
| 123 |
-
"loss":
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
-
"epoch": 0.
|
| 128 |
-
"grad_norm": 1.
|
| 129 |
-
"learning_rate":
|
| 130 |
-
"loss":
|
| 131 |
-
"step": 300
|
| 132 |
-
},
|
| 133 |
-
{
|
| 134 |
-
"epoch": 0.010111223458038422,
|
| 135 |
-
"eval_loss": 5.60944938659668,
|
| 136 |
-
"eval_runtime": 7.4311,
|
| 137 |
-
"eval_samples_per_second": 1282.036,
|
| 138 |
-
"eval_steps_per_second": 0.942,
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
-
"epoch": 0.
|
| 143 |
-
"grad_norm":
|
| 144 |
-
"learning_rate":
|
| 145 |
-
"loss":
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
-
"epoch": 0.
|
| 150 |
-
"grad_norm": 1.
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss":
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
-
"epoch": 0.
|
| 157 |
-
"grad_norm": 1.
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss":
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
-
"epoch": 0.
|
| 164 |
-
"grad_norm": 1.
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss":
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
-
"epoch": 0.
|
| 171 |
-
"grad_norm": 1.
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss":
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
-
"epoch": 0.
|
| 178 |
-
"eval_loss":
|
| 179 |
-
"eval_runtime": 7.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
-
"epoch": 0.
|
| 186 |
-
"grad_norm": 0.
|
| 187 |
-
"learning_rate": 0.
|
| 188 |
-
"loss":
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
-
"epoch": 0.
|
| 193 |
-
"grad_norm": 1.
|
| 194 |
-
"learning_rate": 0.
|
| 195 |
-
"loss":
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
-
"epoch": 0.
|
| 200 |
-
"grad_norm":
|
| 201 |
-
"learning_rate": 0.
|
| 202 |
-
"loss":
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
-
"epoch": 0.
|
| 207 |
-
"grad_norm":
|
| 208 |
-
"learning_rate": 0.
|
| 209 |
-
"loss":
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
-
"epoch": 0.
|
| 214 |
-
"grad_norm": 1.
|
| 215 |
-
"learning_rate": 0.
|
| 216 |
-
"loss":
|
| 217 |
-
"step": 500
|
| 218 |
-
},
|
| 219 |
-
{
|
| 220 |
-
"epoch": 0.016852039096730706,
|
| 221 |
-
"eval_loss": 4.591919898986816,
|
| 222 |
-
"eval_runtime": 7.3895,
|
| 223 |
-
"eval_samples_per_second": 1289.263,
|
| 224 |
-
"eval_steps_per_second": 0.947,
|
| 225 |
"step": 500
|
| 226 |
},
|
| 227 |
{
|
| 228 |
-
"epoch": 0.
|
| 229 |
-
"grad_norm":
|
| 230 |
-
"learning_rate": 0.
|
| 231 |
-
"loss":
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
-
"epoch": 0.
|
| 236 |
-
"grad_norm": 1.
|
| 237 |
-
"learning_rate": 0.
|
| 238 |
-
"loss":
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
-
"epoch": 0.
|
| 243 |
-
"grad_norm": 1.
|
| 244 |
-
"learning_rate": 0.
|
| 245 |
-
"loss":
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
-
"epoch": 0.
|
| 250 |
-
"grad_norm": 1.
|
| 251 |
-
"learning_rate": 0.
|
| 252 |
-
"loss":
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
-
"epoch": 0.
|
| 257 |
-
"grad_norm":
|
| 258 |
-
"learning_rate": 0.
|
| 259 |
-
"loss":
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
-
"epoch": 0.
|
| 264 |
-
"eval_loss":
|
| 265 |
-
"eval_runtime": 7.
|
| 266 |
-
"eval_samples_per_second":
|
| 267 |
-
"eval_steps_per_second": 0.
|
| 268 |
"step": 600
|
| 269 |
},
|
| 270 |
{
|
| 271 |
-
"epoch": 0.
|
| 272 |
-
"grad_norm": 1.
|
| 273 |
-
"learning_rate": 0.
|
| 274 |
-
"loss":
|
| 275 |
"step": 620
|
| 276 |
},
|
| 277 |
{
|
| 278 |
-
"epoch": 0.
|
| 279 |
-
"grad_norm": 1.
|
| 280 |
-
"learning_rate": 0.
|
| 281 |
-
"loss":
|
| 282 |
"step": 640
|
| 283 |
},
|
| 284 |
{
|
| 285 |
-
"epoch": 0.
|
| 286 |
-
"grad_norm":
|
| 287 |
-
"learning_rate": 0.
|
| 288 |
-
"loss": 4.
|
| 289 |
"step": 660
|
| 290 |
},
|
| 291 |
{
|
| 292 |
-
"epoch": 0.
|
| 293 |
-
"grad_norm": 2.
|
| 294 |
-
"learning_rate": 0.
|
| 295 |
-
"loss": 4.
|
| 296 |
"step": 680
|
| 297 |
},
|
| 298 |
{
|
| 299 |
-
"epoch": 0.
|
| 300 |
-
"grad_norm": 1.
|
| 301 |
-
"learning_rate": 0.
|
| 302 |
-
"loss": 4.
|
| 303 |
-
"step": 700
|
| 304 |
-
},
|
| 305 |
-
{
|
| 306 |
-
"epoch": 0.023592854735422986,
|
| 307 |
-
"eval_loss": 4.144440650939941,
|
| 308 |
-
"eval_runtime": 7.5426,
|
| 309 |
-
"eval_samples_per_second": 1263.095,
|
| 310 |
-
"eval_steps_per_second": 0.928,
|
| 311 |
"step": 700
|
| 312 |
},
|
| 313 |
{
|
| 314 |
-
"epoch": 0.
|
| 315 |
-
"grad_norm":
|
| 316 |
-
"learning_rate": 0.
|
| 317 |
-
"loss": 4.
|
| 318 |
"step": 720
|
| 319 |
},
|
| 320 |
{
|
| 321 |
-
"epoch": 0.
|
| 322 |
-
"grad_norm":
|
| 323 |
-
"learning_rate": 0.
|
| 324 |
-
"loss": 4.
|
| 325 |
"step": 740
|
| 326 |
},
|
| 327 |
{
|
| 328 |
-
"epoch": 0.
|
| 329 |
-
"grad_norm": 2.
|
| 330 |
-
"learning_rate": 0.
|
| 331 |
-
"loss": 4.
|
| 332 |
"step": 760
|
| 333 |
},
|
| 334 |
{
|
| 335 |
-
"epoch": 0.
|
| 336 |
-
"grad_norm": 1.
|
| 337 |
-
"learning_rate": 0.
|
| 338 |
-
"loss": 4.
|
| 339 |
"step": 780
|
| 340 |
},
|
| 341 |
{
|
| 342 |
-
"epoch": 0.
|
| 343 |
-
"grad_norm": 1.
|
| 344 |
-
"learning_rate": 0.
|
| 345 |
-
"loss": 4.
|
| 346 |
"step": 800
|
| 347 |
},
|
| 348 |
{
|
| 349 |
-
"epoch": 0.
|
| 350 |
-
"eval_loss": 4.
|
| 351 |
-
"eval_runtime": 7.
|
| 352 |
-
"eval_samples_per_second":
|
| 353 |
-
"eval_steps_per_second": 0.
|
| 354 |
"step": 800
|
| 355 |
},
|
| 356 |
{
|
| 357 |
-
"epoch": 0.
|
| 358 |
-
"grad_norm": 1.
|
| 359 |
-
"learning_rate": 0.
|
| 360 |
-
"loss":
|
| 361 |
"step": 820
|
| 362 |
},
|
| 363 |
{
|
| 364 |
-
"epoch": 0.
|
| 365 |
-
"grad_norm":
|
| 366 |
-
"learning_rate": 0.
|
| 367 |
-
"loss":
|
| 368 |
"step": 840
|
| 369 |
},
|
| 370 |
{
|
| 371 |
-
"epoch": 0.
|
| 372 |
-
"grad_norm": 1.
|
| 373 |
-
"learning_rate": 0.
|
| 374 |
-
"loss": 4.
|
| 375 |
"step": 860
|
| 376 |
},
|
| 377 |
{
|
| 378 |
-
"epoch": 0.
|
| 379 |
-
"grad_norm": 1.
|
| 380 |
-
"learning_rate": 0.
|
| 381 |
-
"loss":
|
| 382 |
"step": 880
|
| 383 |
},
|
| 384 |
{
|
| 385 |
-
"epoch": 0.
|
| 386 |
-
"grad_norm":
|
| 387 |
-
"learning_rate": 0.
|
| 388 |
-
"loss":
|
| 389 |
-
"step": 900
|
| 390 |
-
},
|
| 391 |
-
{
|
| 392 |
-
"epoch": 0.030333670374115267,
|
| 393 |
-
"eval_loss": 3.954482078552246,
|
| 394 |
-
"eval_runtime": 7.9455,
|
| 395 |
-
"eval_samples_per_second": 1199.051,
|
| 396 |
-
"eval_steps_per_second": 0.881,
|
| 397 |
"step": 900
|
| 398 |
}
|
| 399 |
],
|
| 400 |
"logging_steps": 20,
|
| 401 |
-
"max_steps":
|
| 402 |
"num_input_tokens_seen": 0,
|
| 403 |
"num_train_epochs": 1,
|
| 404 |
"save_steps": 100,
|
|
@@ -414,8 +374,8 @@
|
|
| 414 |
"attributes": {}
|
| 415 |
}
|
| 416 |
},
|
| 417 |
-
"total_flos":
|
| 418 |
-
"train_batch_size":
|
| 419 |
"trial_name": null,
|
| 420 |
"trial_params": null
|
| 421 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.015166835187057633,
|
| 6 |
+
"eval_steps": 200,
|
| 7 |
"global_step": 900,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
+
"epoch": 0.0003370407819346141,
|
| 14 |
+
"grad_norm": 1.4140625,
|
| 15 |
+
"learning_rate": 5.7e-06,
|
| 16 |
+
"loss": 8.324227142333985,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
+
"epoch": 0.0006740815638692282,
|
| 21 |
+
"grad_norm": 1.5390625,
|
| 22 |
+
"learning_rate": 1.17e-05,
|
| 23 |
+
"loss": 8.318060302734375,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
+
"epoch": 0.0010111223458038423,
|
| 28 |
+
"grad_norm": 1.609375,
|
| 29 |
+
"learning_rate": 1.7699999999999997e-05,
|
| 30 |
+
"loss": 8.293100738525391,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
+
"epoch": 0.0013481631277384564,
|
| 35 |
+
"grad_norm": 1.7578125,
|
| 36 |
+
"learning_rate": 2.3699999999999997e-05,
|
| 37 |
+
"loss": 8.227334594726562,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
+
"epoch": 0.0016852039096730705,
|
| 42 |
+
"grad_norm": 1.5234375,
|
| 43 |
+
"learning_rate": 2.97e-05,
|
| 44 |
+
"loss": 8.111893463134766,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
+
"epoch": 0.0020222446916076846,
|
| 49 |
+
"grad_norm": 1.3046875,
|
| 50 |
+
"learning_rate": 3.5699999999999994e-05,
|
| 51 |
+
"loss": 7.976696014404297,
|
| 52 |
"step": 120
|
| 53 |
},
|
| 54 |
{
|
| 55 |
+
"epoch": 0.0023592854735422987,
|
| 56 |
+
"grad_norm": 1.296875,
|
| 57 |
+
"learning_rate": 4.17e-05,
|
| 58 |
+
"loss": 7.851339721679688,
|
| 59 |
"step": 140
|
| 60 |
},
|
| 61 |
{
|
| 62 |
+
"epoch": 0.002696326255476913,
|
| 63 |
+
"grad_norm": 1.3046875,
|
| 64 |
+
"learning_rate": 4.7699999999999994e-05,
|
| 65 |
+
"loss": 7.724923706054687,
|
| 66 |
"step": 160
|
| 67 |
},
|
| 68 |
{
|
| 69 |
+
"epoch": 0.003033367037411527,
|
| 70 |
+
"grad_norm": 1.3125,
|
| 71 |
+
"learning_rate": 5.369999999999999e-05,
|
| 72 |
+
"loss": 7.58428726196289,
|
| 73 |
"step": 180
|
| 74 |
},
|
| 75 |
{
|
| 76 |
+
"epoch": 0.003370407819346141,
|
| 77 |
+
"grad_norm": 1.25,
|
| 78 |
+
"learning_rate": 5.97e-05,
|
| 79 |
+
"loss": 7.439914703369141,
|
| 80 |
"step": 200
|
| 81 |
},
|
| 82 |
{
|
| 83 |
+
"epoch": 0.003370407819346141,
|
| 84 |
+
"eval_loss": 7.356490135192871,
|
| 85 |
+
"eval_runtime": 7.5119,
|
| 86 |
+
"eval_samples_per_second": 1268.257,
|
| 87 |
+
"eval_steps_per_second": 0.932,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
+
"epoch": 0.003707448601280755,
|
| 92 |
+
"grad_norm": 1.25,
|
| 93 |
+
"learning_rate": 6.57e-05,
|
| 94 |
+
"loss": 7.283377075195313,
|
| 95 |
"step": 220
|
| 96 |
},
|
| 97 |
{
|
| 98 |
+
"epoch": 0.004044489383215369,
|
| 99 |
+
"grad_norm": 1.25,
|
| 100 |
+
"learning_rate": 7.17e-05,
|
| 101 |
+
"loss": 7.127851104736328,
|
| 102 |
"step": 240
|
| 103 |
},
|
| 104 |
{
|
| 105 |
+
"epoch": 0.004381530165149983,
|
| 106 |
+
"grad_norm": 1.2109375,
|
| 107 |
+
"learning_rate": 7.769999999999999e-05,
|
| 108 |
+
"loss": 6.964313507080078,
|
| 109 |
"step": 260
|
| 110 |
},
|
| 111 |
{
|
| 112 |
+
"epoch": 0.0047185709470845974,
|
| 113 |
+
"grad_norm": 1.171875,
|
| 114 |
+
"learning_rate": 8.37e-05,
|
| 115 |
+
"loss": 6.807338714599609,
|
| 116 |
"step": 280
|
| 117 |
},
|
| 118 |
{
|
| 119 |
+
"epoch": 0.005055611729019211,
|
| 120 |
+
"grad_norm": 1.1328125,
|
| 121 |
+
"learning_rate": 8.969999999999998e-05,
|
| 122 |
+
"loss": 6.667655181884766,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 123 |
"step": 300
|
| 124 |
},
|
| 125 |
{
|
| 126 |
+
"epoch": 0.005392652510953826,
|
| 127 |
+
"grad_norm": 1.109375,
|
| 128 |
+
"learning_rate": 9.57e-05,
|
| 129 |
+
"loss": 6.523377227783203,
|
| 130 |
"step": 320
|
| 131 |
},
|
| 132 |
{
|
| 133 |
+
"epoch": 0.005729693292888439,
|
| 134 |
+
"grad_norm": 1.1171875,
|
| 135 |
+
"learning_rate": 0.00010169999999999999,
|
| 136 |
+
"loss": 6.383005142211914,
|
| 137 |
"step": 340
|
| 138 |
},
|
| 139 |
{
|
| 140 |
+
"epoch": 0.006066734074823054,
|
| 141 |
+
"grad_norm": 1.6875,
|
| 142 |
+
"learning_rate": 0.00010769999999999999,
|
| 143 |
+
"loss": 6.261091232299805,
|
| 144 |
"step": 360
|
| 145 |
},
|
| 146 |
{
|
| 147 |
+
"epoch": 0.0064037748567576675,
|
| 148 |
+
"grad_norm": 1.140625,
|
| 149 |
+
"learning_rate": 0.00011369999999999999,
|
| 150 |
+
"loss": 6.122833251953125,
|
| 151 |
"step": 380
|
| 152 |
},
|
| 153 |
{
|
| 154 |
+
"epoch": 0.006740815638692282,
|
| 155 |
+
"grad_norm": 1.3984375,
|
| 156 |
+
"learning_rate": 0.0001197,
|
| 157 |
+
"loss": 6.019657897949219,
|
| 158 |
"step": 400
|
| 159 |
},
|
| 160 |
{
|
| 161 |
+
"epoch": 0.006740815638692282,
|
| 162 |
+
"eval_loss": 5.966014385223389,
|
| 163 |
+
"eval_runtime": 7.516,
|
| 164 |
+
"eval_samples_per_second": 1267.556,
|
| 165 |
+
"eval_steps_per_second": 0.931,
|
| 166 |
"step": 400
|
| 167 |
},
|
| 168 |
{
|
| 169 |
+
"epoch": 0.007077856420626896,
|
| 170 |
+
"grad_norm": 0.98046875,
|
| 171 |
+
"learning_rate": 0.0001257,
|
| 172 |
+
"loss": 5.9373779296875,
|
| 173 |
"step": 420
|
| 174 |
},
|
| 175 |
{
|
| 176 |
+
"epoch": 0.00741489720256151,
|
| 177 |
+
"grad_norm": 1.6328125,
|
| 178 |
+
"learning_rate": 0.00013169999999999998,
|
| 179 |
+
"loss": 5.839211273193359,
|
| 180 |
"step": 440
|
| 181 |
},
|
| 182 |
{
|
| 183 |
+
"epoch": 0.007751937984496124,
|
| 184 |
+
"grad_norm": 0.9609375,
|
| 185 |
+
"learning_rate": 0.00013769999999999999,
|
| 186 |
+
"loss": 5.740922927856445,
|
| 187 |
"step": 460
|
| 188 |
},
|
| 189 |
{
|
| 190 |
+
"epoch": 0.008088978766430738,
|
| 191 |
+
"grad_norm": 0.90234375,
|
| 192 |
+
"learning_rate": 0.00014369999999999997,
|
| 193 |
+
"loss": 5.6399181365966795,
|
| 194 |
"step": 480
|
| 195 |
},
|
| 196 |
{
|
| 197 |
+
"epoch": 0.008426019548365353,
|
| 198 |
+
"grad_norm": 1.1796875,
|
| 199 |
+
"learning_rate": 0.00014969999999999998,
|
| 200 |
+
"loss": 5.560699081420898,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 201 |
"step": 500
|
| 202 |
},
|
| 203 |
{
|
| 204 |
+
"epoch": 0.008763060330299966,
|
| 205 |
+
"grad_norm": 2.328125,
|
| 206 |
+
"learning_rate": 0.0001557,
|
| 207 |
+
"loss": 5.474863433837891,
|
| 208 |
"step": 520
|
| 209 |
},
|
| 210 |
{
|
| 211 |
+
"epoch": 0.00910010111223458,
|
| 212 |
+
"grad_norm": 1.125,
|
| 213 |
+
"learning_rate": 0.0001617,
|
| 214 |
+
"loss": 5.396588516235352,
|
| 215 |
"step": 540
|
| 216 |
},
|
| 217 |
{
|
| 218 |
+
"epoch": 0.009437141894169195,
|
| 219 |
+
"grad_norm": 1.6484375,
|
| 220 |
+
"learning_rate": 0.0001677,
|
| 221 |
+
"loss": 5.331023406982422,
|
| 222 |
"step": 560
|
| 223 |
},
|
| 224 |
{
|
| 225 |
+
"epoch": 0.00977418267610381,
|
| 226 |
+
"grad_norm": 1.0703125,
|
| 227 |
+
"learning_rate": 0.00017369999999999997,
|
| 228 |
+
"loss": 5.257175445556641,
|
| 229 |
"step": 580
|
| 230 |
},
|
| 231 |
{
|
| 232 |
+
"epoch": 0.010111223458038422,
|
| 233 |
+
"grad_norm": 2.359375,
|
| 234 |
+
"learning_rate": 0.00017969999999999998,
|
| 235 |
+
"loss": 5.152382659912109,
|
| 236 |
"step": 600
|
| 237 |
},
|
| 238 |
{
|
| 239 |
+
"epoch": 0.010111223458038422,
|
| 240 |
+
"eval_loss": 5.1175713539123535,
|
| 241 |
+
"eval_runtime": 7.457,
|
| 242 |
+
"eval_samples_per_second": 1277.585,
|
| 243 |
+
"eval_steps_per_second": 0.939,
|
| 244 |
"step": 600
|
| 245 |
},
|
| 246 |
{
|
| 247 |
+
"epoch": 0.010448264239973037,
|
| 248 |
+
"grad_norm": 1.375,
|
| 249 |
+
"learning_rate": 0.0001857,
|
| 250 |
+
"loss": 5.096985244750977,
|
| 251 |
"step": 620
|
| 252 |
},
|
| 253 |
{
|
| 254 |
+
"epoch": 0.010785305021907651,
|
| 255 |
+
"grad_norm": 1.421875,
|
| 256 |
+
"learning_rate": 0.0001917,
|
| 257 |
+
"loss": 5.019801330566406,
|
| 258 |
"step": 640
|
| 259 |
},
|
| 260 |
{
|
| 261 |
+
"epoch": 0.011122345803842264,
|
| 262 |
+
"grad_norm": 2.125,
|
| 263 |
+
"learning_rate": 0.00019769999999999998,
|
| 264 |
+
"loss": 4.979957962036133,
|
| 265 |
"step": 660
|
| 266 |
},
|
| 267 |
{
|
| 268 |
+
"epoch": 0.011459386585776879,
|
| 269 |
+
"grad_norm": 2.203125,
|
| 270 |
+
"learning_rate": 0.0002037,
|
| 271 |
+
"loss": 4.945652008056641,
|
| 272 |
"step": 680
|
| 273 |
},
|
| 274 |
{
|
| 275 |
+
"epoch": 0.011796427367711493,
|
| 276 |
+
"grad_norm": 1.8515625,
|
| 277 |
+
"learning_rate": 0.00020969999999999997,
|
| 278 |
+
"loss": 4.866051483154297,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 279 |
"step": 700
|
| 280 |
},
|
| 281 |
{
|
| 282 |
+
"epoch": 0.012133468149646108,
|
| 283 |
+
"grad_norm": 1.625,
|
| 284 |
+
"learning_rate": 0.00021569999999999998,
|
| 285 |
+
"loss": 4.841766357421875,
|
| 286 |
"step": 720
|
| 287 |
},
|
| 288 |
{
|
| 289 |
+
"epoch": 0.01247050893158072,
|
| 290 |
+
"grad_norm": 3.40625,
|
| 291 |
+
"learning_rate": 0.00022169999999999997,
|
| 292 |
+
"loss": 4.798672103881836,
|
| 293 |
"step": 740
|
| 294 |
},
|
| 295 |
{
|
| 296 |
+
"epoch": 0.012807549713515335,
|
| 297 |
+
"grad_norm": 2.21875,
|
| 298 |
+
"learning_rate": 0.00022769999999999998,
|
| 299 |
+
"loss": 4.768531036376953,
|
| 300 |
"step": 760
|
| 301 |
},
|
| 302 |
{
|
| 303 |
+
"epoch": 0.01314459049544995,
|
| 304 |
+
"grad_norm": 1.640625,
|
| 305 |
+
"learning_rate": 0.0002337,
|
| 306 |
+
"loss": 4.73585319519043,
|
| 307 |
"step": 780
|
| 308 |
},
|
| 309 |
{
|
| 310 |
+
"epoch": 0.013481631277384564,
|
| 311 |
+
"grad_norm": 1.6875,
|
| 312 |
+
"learning_rate": 0.0002397,
|
| 313 |
+
"loss": 4.697021865844727,
|
| 314 |
"step": 800
|
| 315 |
},
|
| 316 |
{
|
| 317 |
+
"epoch": 0.013481631277384564,
|
| 318 |
+
"eval_loss": 4.676848411560059,
|
| 319 |
+
"eval_runtime": 7.4394,
|
| 320 |
+
"eval_samples_per_second": 1280.62,
|
| 321 |
+
"eval_steps_per_second": 0.941,
|
| 322 |
"step": 800
|
| 323 |
},
|
| 324 |
{
|
| 325 |
+
"epoch": 0.013818672059319177,
|
| 326 |
+
"grad_norm": 1.6484375,
|
| 327 |
+
"learning_rate": 0.00024569999999999995,
|
| 328 |
+
"loss": 4.658943176269531,
|
| 329 |
"step": 820
|
| 330 |
},
|
| 331 |
{
|
| 332 |
+
"epoch": 0.014155712841253791,
|
| 333 |
+
"grad_norm": 3.234375,
|
| 334 |
+
"learning_rate": 0.0002517,
|
| 335 |
+
"loss": 4.5957294464111325,
|
| 336 |
"step": 840
|
| 337 |
},
|
| 338 |
{
|
| 339 |
+
"epoch": 0.014492753623188406,
|
| 340 |
+
"grad_norm": 1.578125,
|
| 341 |
+
"learning_rate": 0.0002577,
|
| 342 |
+
"loss": 4.5967552185058596,
|
| 343 |
"step": 860
|
| 344 |
},
|
| 345 |
{
|
| 346 |
+
"epoch": 0.01482979440512302,
|
| 347 |
+
"grad_norm": 1.3203125,
|
| 348 |
+
"learning_rate": 0.00026369999999999996,
|
| 349 |
+
"loss": 4.549871444702148,
|
| 350 |
"step": 880
|
| 351 |
},
|
| 352 |
{
|
| 353 |
+
"epoch": 0.015166835187057633,
|
| 354 |
+
"grad_norm": 2.28125,
|
| 355 |
+
"learning_rate": 0.0002697,
|
| 356 |
+
"loss": 4.512709045410157,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 357 |
"step": 900
|
| 358 |
}
|
| 359 |
],
|
| 360 |
"logging_steps": 20,
|
| 361 |
+
"max_steps": 5000,
|
| 362 |
"num_input_tokens_seen": 0,
|
| 363 |
"num_train_epochs": 1,
|
| 364 |
"save_steps": 100,
|
|
|
|
| 374 |
"attributes": {}
|
| 375 |
}
|
| 376 |
},
|
| 377 |
+
"total_flos": 21782908108800.0,
|
| 378 |
+
"train_batch_size": 16,
|
| 379 |
"trial_name": null,
|
| 380 |
"trial_params": null
|
| 381 |
}
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c8ba4204aa09d2b6d0fe9a4a91b258d44c21a6738a757711b7ef84256d0583a1
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-3L_run/training_log.jsonl
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
zain/Activation/wandb/debug-internal.log
CHANGED
|
@@ -1,63 +1,35 @@
|
|
| 1 |
-
{"time":"2026-08-
|
| 2 |
-
{"time":"2026-08-
|
| 3 |
-
{"time":"2026-08-
|
| 4 |
-
{"time":"2026-08-
|
| 5 |
-
{"time":"2026-08-
|
| 6 |
-
{"time":"2026-08-
|
| 7 |
-
{"time":"2026-08-
|
| 8 |
-
{"time":"2026-08-
|
| 9 |
-
{"time":"2026-08-
|
| 10 |
-
{"time":"2026-08-
|
| 11 |
-
{"time":"2026-08-
|
| 12 |
-
{"time":"2026-08-
|
| 13 |
-
{"time":"2026-08-
|
| 14 |
-
{"time":"2026-08-
|
| 15 |
-
{"time":"2026-08-
|
| 16 |
-
{"time":"2026-08-
|
| 17 |
-
{"time":"2026-08-
|
| 18 |
-
{"time":"2026-08-
|
| 19 |
-
{"time":"2026-08-
|
| 20 |
-
{"time":"2026-08-
|
| 21 |
-
{"time":"2026-08-
|
| 22 |
-
{"time":"2026-08-
|
| 23 |
-
{"time":"2026-08-
|
| 24 |
-
{"time":"2026-08-
|
| 25 |
-
{"time":"2026-08-
|
| 26 |
-
{"time":"2026-08-
|
| 27 |
-
{"time":"2026-08-
|
| 28 |
-
{"time":"2026-08-
|
| 29 |
-
{"time":"2026-08-
|
| 30 |
-
{"time":"2026-08-
|
| 31 |
-
{"time":"2026-08-
|
| 32 |
-
{"time":"2026-08-
|
| 33 |
-
{"time":"2026-08-
|
| 34 |
-
{"time":"2026-08-
|
| 35 |
-
{"time":"2026-08-
|
| 36 |
-
{"time":"2026-08-19T00:09:51.992822127Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":89,"history_lines":6,"events_offset":25,"events_lines":2,"console_offset":154,"console_lines":1}
|
| 37 |
-
{"time":"2026-08-19T00:09:52.109096249Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 38 |
-
{"time":"2026-08-19T00:10:06.992523849Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":95,"history_lines":6,"events_offset":27,"events_lines":2,"console_offset":160,"console_lines":23}
|
| 39 |
-
{"time":"2026-08-19T00:10:07.154689008Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 40 |
-
{"time":"2026-08-19T00:10:21.993355847Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":101,"history_lines":6,"events_offset":29,"events_lines":2,"console_offset":176,"console_lines":1}
|
| 41 |
-
{"time":"2026-08-19T00:10:22.178839986Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 42 |
-
{"time":"2026-08-19T00:10:36.992657682Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":107,"history_lines":6,"events_offset":31,"events_lines":2,"console_offset":182,"console_lines":23}
|
| 43 |
-
{"time":"2026-08-19T00:10:37.1185939Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 44 |
-
{"time":"2026-08-19T00:10:51.993057513Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":113,"history_lines":7,"events_offset":33,"events_lines":2,"console_offset":198,"console_lines":1}
|
| 45 |
-
{"time":"2026-08-19T00:10:52.101588155Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 46 |
-
{"time":"2026-08-19T00:11:06.992973349Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":120,"history_lines":7,"events_offset":35,"events_lines":2,"console_offset":204,"console_lines":29}
|
| 47 |
-
{"time":"2026-08-19T00:11:07.134793732Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 48 |
-
{"time":"2026-08-19T00:11:21.994425336Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":127,"history_lines":8,"events_offset":37,"events_lines":2,"console_offset":231,"console_lines":1}
|
| 49 |
-
{"time":"2026-08-19T00:11:22.134821113Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 50 |
-
{"time":"2026-08-19T00:11:36.993054503Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":135,"history_lines":7,"events_offset":39,"events_lines":2,"console_offset":233,"console_lines":25}
|
| 51 |
-
{"time":"2026-08-19T00:11:37.111533939Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 52 |
-
{"time":"2026-08-19T00:11:51.992604078Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":142,"history_lines":7,"events_offset":41,"events_lines":2,"console_offset":253,"console_lines":1}
|
| 53 |
-
{"time":"2026-08-19T00:11:52.115538255Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 54 |
-
{"time":"2026-08-19T00:12:06.992994064Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":149,"history_lines":2,"events_offset":43,"events_lines":2,"console_offset":258,"console_lines":20}
|
| 55 |
-
{"time":"2026-08-19T00:12:07.157903259Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 56 |
-
{"time":"2026-08-19T00:12:08.630011645Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
|
| 57 |
-
{"time":"2026-08-19T00:12:08.630264558Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":151,"history_lines":1,"events_offset":45,"events_lines":1,"console_offset":277,"console_lines":7,"uploaded_len":3,"complete":true,"exit_code":0}
|
| 58 |
-
{"time":"2026-08-19T00:12:08.908163631Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 59 |
-
{"time":"2026-08-19T00:12:08.909177309Z","level":"INFO","msg":"handler: operation stats","stats":{}}
|
| 60 |
-
{"time":"2026-08-19T00:12:08.911408845Z","level":"INFO","msg":"stream: finishing up"}
|
| 61 |
-
{"time":"2026-08-19T00:12:08.911426034Z","level":"INFO","msg":"handler: closed"}
|
| 62 |
-
{"time":"2026-08-19T00:12:08.911484689Z","level":"INFO","msg":"sender: closed"}
|
| 63 |
-
{"time":"2026-08-19T00:12:08.911488467Z","level":"INFO","msg":"stream: all finished"}
|
|
|
|
| 1 |
+
{"time":"2026-08-19T16:30:14.408862805Z","level":"INFO","msg":"wandb-core"}
|
| 2 |
+
{"time":"2026-08-19T16:30:14.409145225Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"}
|
| 3 |
+
{"time":"2026-08-19T16:30:14.673839961Z","level":"INFO","msg":"stream: created new stream","id":"smtfejrp"}
|
| 4 |
+
{"time":"2026-08-19T16:30:14.673914385Z","level":"INFO","msg":"handler: started"}
|
| 5 |
+
{"time":"2026-08-19T16:30:14.674023743Z","level":"INFO","msg":"stream: started"}
|
| 6 |
+
{"time":"2026-08-19T16:30:14.67404595Z","level":"INFO","msg":"writer: started","stream_id":"smtfejrp"}
|
| 7 |
+
{"time":"2026-08-19T16:30:14.674053413Z","level":"INFO","msg":"sender: started"}
|
| 8 |
+
{"time":"2026-08-19T16:30:15.573430136Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1}
|
| 9 |
+
{"time":"2026-08-19T16:30:15.803201108Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 10 |
+
{"time":"2026-08-19T16:30:30.574169941Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":16,"events_offset":0,"events_lines":1,"console_offset":0,"console_lines":34,"uploaded_len":2}
|
| 11 |
+
{"time":"2026-08-19T16:30:30.711458871Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 12 |
+
{"time":"2026-08-19T16:30:45.574202013Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":16,"history_lines":16,"events_offset":1,"events_lines":2,"console_offset":33,"console_lines":27}
|
| 13 |
+
{"time":"2026-08-19T16:30:45.755515219Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 14 |
+
{"time":"2026-08-19T16:31:00.573737519Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":32,"history_lines":11,"events_offset":3,"events_lines":2,"console_offset":54,"console_lines":1}
|
| 15 |
+
{"time":"2026-08-19T16:31:00.75827199Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 16 |
+
{"time":"2026-08-19T16:31:15.574343371Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":43,"history_lines":11,"events_offset":5,"events_lines":2,"console_offset":60,"console_lines":43}
|
| 17 |
+
{"time":"2026-08-19T16:31:15.798424562Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 18 |
+
{"time":"2026-08-19T16:31:30.573773942Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":54,"history_lines":18,"events_offset":7,"events_lines":2,"console_offset":96,"console_lines":1}
|
| 19 |
+
{"time":"2026-08-19T16:31:30.694320256Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 20 |
+
{"time":"2026-08-19T16:31:45.574312952Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":72,"history_lines":15,"events_offset":9,"events_lines":2,"console_offset":102,"console_lines":63}
|
| 21 |
+
{"time":"2026-08-19T16:31:45.774027402Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 22 |
+
{"time":"2026-08-19T16:32:00.574639142Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":87,"history_lines":11,"events_offset":11,"events_lines":2,"console_offset":159,"console_lines":1}
|
| 23 |
+
{"time":"2026-08-19T16:32:00.704830872Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 24 |
+
{"time":"2026-08-19T16:32:15.573881284Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":98,"history_lines":11,"events_offset":13,"events_lines":2,"console_offset":165,"console_lines":43}
|
| 25 |
+
{"time":"2026-08-19T16:32:15.785173455Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 26 |
+
{"time":"2026-08-19T16:32:30.574343755Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":109,"history_lines":19,"events_offset":15,"events_lines":2,"console_offset":201,"console_lines":1}
|
| 27 |
+
{"time":"2026-08-19T16:32:30.715820396Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 28 |
+
{"time":"2026-08-19T16:32:45.573977636Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":128,"history_lines":14,"events_offset":17,"events_lines":2,"console_offset":207,"console_lines":63}
|
| 29 |
+
{"time":"2026-08-19T16:32:45.688687126Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 30 |
+
{"time":"2026-08-19T16:33:00.574401218Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":142,"history_lines":11,"events_offset":19,"events_lines":2,"console_offset":264,"console_lines":1}
|
| 31 |
+
{"time":"2026-08-19T16:33:00.726033404Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 32 |
+
{"time":"2026-08-19T16:33:15.575478736Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":153,"history_lines":13,"events_offset":21,"events_lines":2,"console_offset":270,"console_lines":49}
|
| 33 |
+
{"time":"2026-08-19T16:33:15.746742621Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 34 |
+
{"time":"2026-08-19T16:33:30.574073819Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":166,"history_lines":18,"events_offset":23,"events_lines":2,"console_offset":317,"console_lines":1}
|
| 35 |
+
{"time":"2026-08-19T16:33:30.725976627Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
zain/Activation/wandb/debug.log
CHANGED
|
@@ -1,25 +1,23 @@
|
|
| 1 |
-
2026-08-19
|
| 2 |
-
2026-08-19
|
| 3 |
-
2026-08-19
|
| 4 |
-
2026-08-19
|
|
|
|
|
|
|
|
|
|
| 5 |
config: {'_wandb': {}}
|
| 6 |
-
2026-08-19
|
| 7 |
-
2026-08-19
|
| 8 |
-
2026-08-19
|
| 9 |
-
2026-08-19
|
| 10 |
-
2026-08-19
|
| 11 |
-
2026-08-19
|
| 12 |
-
2026-08-19
|
| 13 |
-
2026-08-19
|
| 14 |
-
2026-08-19
|
| 15 |
-
2026-08-19
|
| 16 |
-
2026-08-19
|
| 17 |
-
2026-08-19
|
| 18 |
-
2026-08-19
|
| 19 |
-
2026-08-19
|
| 20 |
-
2026-08-19
|
| 21 |
-
2026-08-19 00:12:08,222 INFO MainThread:3613580 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/research-ultimate/6zbmve2d
|
| 22 |
-
2026-08-19 00:12:08,222 INFO MainThread:3613580 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0
|
| 23 |
-
2026-08-19 00:12:08,222 INFO MainThread:3613580 [wandb_run.py:_restore():2570] restore
|
| 24 |
-
2026-08-19 00:12:08,222 INFO MainThread:3613580 [wandb_run.py:_restore():2576] restore done
|
| 25 |
-
2026-08-19 00:12:08,910 INFO MainThread:3613580 [wandb_run.py:_footer_sync_info():3993] logging synced files
|
|
|
|
| 1 |
+
2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1
|
| 2 |
+
2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_setup.py:_flush():81] Configure stats pid to 3508170
|
| 3 |
+
2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_setup.py:_flush():81] Loading settings from environment variables
|
| 4 |
+
2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260819_163014-smtfejrp/logs/debug.log
|
| 5 |
+
2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260819_163014-smtfejrp/logs/debug-internal.log
|
| 6 |
+
2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:init():772] calling init triggers
|
| 7 |
+
2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:init():777] wandb.init called with sweep_config: {}
|
| 8 |
config: {'_wandb': {}}
|
| 9 |
+
2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:init():820] starting backend
|
| 10 |
+
2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:init():826] Connected to an existing wandb-core service via WANDB_SERVICE
|
| 11 |
+
2026-08-19 16:30:14,407 INFO MainThread:3508170 [wandb_init.py:init():835] sending inform_init request
|
| 12 |
+
2026-08-19 16:30:14,674 INFO MainThread:3508170 [wandb_init.py:init():840] backend started and connected
|
| 13 |
+
2026-08-19 16:30:14,678 INFO MainThread:3508170 [wandb_init.py:init():910] updated telemetry
|
| 14 |
+
2026-08-19 16:30:14,685 INFO MainThread:3508170 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout
|
| 15 |
+
2026-08-19 16:30:15,492 INFO MainThread:3508170 [wandb_init.py:init():978] starting run threads in backend
|
| 16 |
+
2026-08-19 16:30:15,566 INFO MainThread:3508170 [wandb_run.py:_console_start():2621] atexit reg
|
| 17 |
+
2026-08-19 16:30:15,566 INFO MainThread:3508170 [wandb_run.py:_redirect():2471] redirect: wrap_raw
|
| 18 |
+
2026-08-19 16:30:15,566 INFO MainThread:3508170 [wandb_run.py:_redirect():2540] Wrapping output streams.
|
| 19 |
+
2026-08-19 16:30:15,566 INFO MainThread:3508170 [wandb_run.py:_redirect():2563] Redirects installed.
|
| 20 |
+
2026-08-19 16:30:15,569 INFO MainThread:3508170 [wandb_init.py:init():1016] run started, returning control to user process
|
| 21 |
+
2026-08-19 16:30:15,570 INFO MainThread:3508170 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 3, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'linear', 'waleed_beta': 10.0, 'powlu_m': 3.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/mlp-linear-3L_run', 'per_device_train_batch_size': 16, 'num_train_epochs': 1, 'max_steps': 5000, 'learning_rate': 0.0003, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 1000, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-linear-3L-1.0M-20260819-163013', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 200, 'eval_delay': 0, 'per_device_eval_batch_size': 1500, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-mlp-linear-3L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
|
| 22 |
+
2026-08-19 16:30:15,572 INFO MainThread:3508170 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 1016704 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x1495a6a04a90>>
|
| 23 |
+
2026-08-19 16:30:15,572 INFO MainThread:3508170 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 1016704 None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
zain/Activation/wandb/run-20260819_163014-smtfejrp/files/output.log
ADDED
|
@@ -0,0 +1,364 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
0%| | 0/5000 [00:00<?, ?it/s][transformers] `use_return_dict` is deprecated! Use `return_dict` instead!
|
| 2 |
+
[INFO] Causal mask (float with -inf) applied to all attention layers.
|
| 3 |
+
2%|▏ | 100/5000 [00:03<01:50, 44.29it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 4 |
+
{'loss': '8.324', 'grad_norm': '1.414', 'learning_rate': '5.7e-06', 'epoch': '0.000337', 'train/total_time_seconds': '0.7989', 'train/time_per_step_avg': '0.03995', 'train/epoch_time_elapsed': '1.159', 'train/estimated_remaining_minutes': '3.315'}
|
| 5 |
+
{'loss': '8.318', 'grad_norm': '1.539', 'learning_rate': '1.17e-05', 'epoch': '0.0006741', 'train/total_time_seconds': '1.003', 'train/time_per_step_avg': '0.02508', 'train/epoch_time_elapsed': '1.621', 'train/estimated_remaining_minutes': '2.073'}
|
| 6 |
+
{'loss': '8.293', 'grad_norm': '1.609', 'learning_rate': '1.77e-05', 'epoch': '0.001011', 'train/total_time_seconds': '1.202', 'train/time_per_step_avg': '0.02003', 'train/epoch_time_elapsed': '2.112', 'train/estimated_remaining_minutes': '1.649'}
|
| 7 |
+
{'loss': '8.227', 'grad_norm': '1.758', 'learning_rate': '2.37e-05', 'epoch': '0.001348', 'train/total_time_seconds': '1.393', 'train/time_per_step_avg': '0.01741', 'train/epoch_time_elapsed': '2.597', 'train/estimated_remaining_minutes': '1.428'}
|
| 8 |
+
{'loss': '8.112', 'grad_norm': '1.523', 'learning_rate': '2.97e-05', 'epoch': '0.001685', 'train/total_time_seconds': '1.571', 'train/time_per_step_avg': '0.01571', 'train/epoch_time_elapsed': '3.029', 'train/estimated_remaining_minutes': '1.283'}
|
| 9 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 10 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 11 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 12 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 227.99it/s]
|
| 13 |
+
4%|▍ | 200/5000 [00:12<01:44, 46.00[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 14 |
+
{'loss': '7.977', 'grad_norm': '1.305', 'learning_rate': '3.57e-05', 'epoch': '0.002022', 'train/total_time_seconds': '1.748', 'train/time_per_step_avg': '0.00949', 'train/epoch_time_elapsed': '3.481', 'train/estimated_remaining_minutes': '1.185'}
|
| 15 |
+
{'loss': '7.851', 'grad_norm': '1.297', 'learning_rate': '4.17e-05', 'epoch': '0.002359', 'train/total_time_seconds': '1.929', 'train/time_per_step_avg': '0.009256', 'train/epoch_time_elapsed': '3.92', 'train/estimated_remaining_minutes': '1.116'}
|
| 16 |
+
{'loss': '7.725', 'grad_norm': '1.305', 'learning_rate': '4.77e-05', 'epoch': '0.002696', 'train/total_time_seconds': '2.105', 'train/time_per_step_avg': '0.009031', 'train/epoch_time_elapsed': '4.349', 'train/estimated_remaining_minutes': '1.061'}
|
| 17 |
+
{'loss': '7.584', 'grad_norm': '1.312', 'learning_rate': '5.37e-05', 'epoch': '0.003033', 'train/total_time_seconds': '2.288', 'train/time_per_step_avg': '0.008948', 'train/epoch_time_elapsed': '4.788', 'train/estimated_remaining_minutes': '1.021'}
|
| 18 |
+
{'loss': '7.44', 'grad_norm': '1.25', 'learning_rate': '5.97e-05', 'epoch': '0.00337', 'train/total_time_seconds': '2.468', 'train/time_per_step_avg': '0.008976', 'train/epoch_time_elapsed': '5.223', 'train/estimated_remaining_minutes': '0.9873'}
|
| 19 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 20 |
+
{'eval_loss': '7.356', 'eval_runtime': '7.512', 'eval_samples_per_second': '1268', 'eval_steps_per_second': '0.932', 'epoch': '0.00337', 'train/total_time_seconds': '2.468', 'train/time_per_step_avg': '0.008976', 'train/epoch_time_elapsed': '12.74', 'train/estimated_remaining_minutes': '0.9873'}
|
| 21 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 22 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 23 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 359.32it/s]
|
| 24 |
+
6%|▌ | 300/5000 [00:14<01:43, 45.61it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 25 |
+
{'loss': '7.283', 'grad_norm': '1.25', 'learning_rate': '6.57e-05', 'epoch': '0.003707', 'train/total_time_seconds': '2.648', 'train/time_per_step_avg': '0.008997', 'train/epoch_time_elapsed': '13.19', 'train/estimated_remaining_minutes': '0.9588'}
|
| 26 |
+
{'loss': '7.128', 'grad_norm': '1.25', 'learning_rate': '7.17e-05', 'epoch': '0.004044', 'train/total_time_seconds': '2.823', 'train/time_per_step_avg': '0.008943', 'train/epoch_time_elapsed': '13.62', 'train/estimated_remaining_minutes': '0.9332'}
|
| 27 |
+
{'loss': '6.964', 'grad_norm': '1.211', 'learning_rate': '7.77e-05', 'epoch': '0.004382', 'train/total_time_seconds': '2.998', 'train/time_per_step_avg': '0.008929', 'train/epoch_time_elapsed': '14.05', 'train/estimated_remaining_minutes': '0.9108'}
|
| 28 |
+
{'loss': '6.807', 'grad_norm': '1.172', 'learning_rate': '8.37e-05', 'epoch': '0.004719', 'train/total_time_seconds': '3.173', 'train/time_per_step_avg': '0.008853', 'train/epoch_time_elapsed': '14.48', 'train/estimated_remaining_minutes': '0.8915'}
|
| 29 |
+
{'loss': '6.668', 'grad_norm': '1.133', 'learning_rate': '8.97e-05', 'epoch': '0.005056', 'train/total_time_seconds': '3.348', 'train/time_per_step_avg': '0.008801', 'train/epoch_time_elapsed': '14.91', 'train/estimated_remaining_minutes': '0.8743'}
|
| 30 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 31 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 32 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 33 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 360.06it/s]
|
| 34 |
+
8%|▊ | 400/5000 [00:24<01:46, 43.02[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 35 |
+
{'loss': '6.523', 'grad_norm': '1.109', 'learning_rate': '9.57e-05', 'epoch': '0.005393', 'train/total_time_seconds': '3.531', 'train/time_per_step_avg': '0.008839', 'train/epoch_time_elapsed': '15.38', 'train/estimated_remaining_minutes': '0.8608'}
|
| 36 |
+
{'loss': '6.383', 'grad_norm': '1.117', 'learning_rate': '0.0001017', 'epoch': '0.00573', 'train/total_time_seconds': '3.71', 'train/time_per_step_avg': '0.008873', 'train/epoch_time_elapsed': '15.81', 'train/estimated_remaining_minutes': '0.8475'}
|
| 37 |
+
{'loss': '6.261', 'grad_norm': '1.688', 'learning_rate': '0.0001077', 'epoch': '0.006067', 'train/total_time_seconds': '3.889', 'train/time_per_step_avg': '0.008913', 'train/epoch_time_elapsed': '16.24', 'train/estimated_remaining_minutes': '0.8354'}
|
| 38 |
+
{'loss': '6.123', 'grad_norm': '1.141', 'learning_rate': '0.0001137', 'epoch': '0.006404', 'train/total_time_seconds': '4.09', 'train/time_per_step_avg': '0.009175', 'train/epoch_time_elapsed': '16.73', 'train/estimated_remaining_minutes': '0.8289'}
|
| 39 |
+
{'loss': '6.02', 'grad_norm': '1.398', 'learning_rate': '0.0001197', 'epoch': '0.006741', 'train/total_time_seconds': '4.287', 'train/time_per_step_avg': '0.009382', 'train/epoch_time_elapsed': '17.18', 'train/estimated_remaining_minutes': '0.8216'}
|
| 40 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 41 |
+
{'eval_loss': '5.966', 'eval_runtime': '7.516', 'eval_samples_per_second': '1268', 'eval_steps_per_second': '0.931', 'epoch': '0.006741', 'train/total_time_seconds': '4.287', 'train/time_per_step_avg': '0.009382', 'train/epoch_time_elapsed': '24.7', 'train/estimated_remaining_minutes': '0.8216'}
|
| 42 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 43 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 44 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 337.95it/s]
|
| 45 |
+
10%|█ | 500/5000 [00:27<01:55, 39.07it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 46 |
+
{'loss': '5.937', 'grad_norm': '0.9805', 'learning_rate': '0.0001257', 'epoch': '0.007078', 'train/total_time_seconds': '4.473', 'train/time_per_step_avg': '0.00942', 'train/epoch_time_elapsed': '25.17', 'train/estimated_remaining_minutes': '0.813'}
|
| 47 |
+
{'loss': '5.839', 'grad_norm': '1.633', 'learning_rate': '0.0001317', 'epoch': '0.007415', 'train/total_time_seconds': '4.662', 'train/time_per_step_avg': '0.009518', 'train/epoch_time_elapsed': '25.62', 'train/estimated_remaining_minutes': '0.8053'}
|
| 48 |
+
{'loss': '5.741', 'grad_norm': '0.9609', 'learning_rate': '0.0001377', 'epoch': '0.007752', 'train/total_time_seconds': '4.846', 'train/time_per_step_avg': '0.009575', 'train/epoch_time_elapsed': '26.06', 'train/estimated_remaining_minutes': '0.7972'}
|
| 49 |
+
{'loss': '5.64', 'grad_norm': '0.9023', 'learning_rate': '0.0001437', 'epoch': '0.008089', 'train/total_time_seconds': '5.032', 'train/time_per_step_avg': '0.009414', 'train/epoch_time_elapsed': '26.5', 'train/estimated_remaining_minutes': '0.7897'}
|
| 50 |
+
{'loss': '5.561', 'grad_norm': '1.18', 'learning_rate': '0.0001497', 'epoch': '0.008426', 'train/total_time_seconds': '5.24', 'train/time_per_step_avg': '0.009536', 'train/epoch_time_elapsed': '27', 'train/estimated_remaining_minutes': '0.786'}
|
| 51 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 52 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 53 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 54 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 107.02it/s]
|
| 55 |
+
12%|█▏ | 600/5000 [00:36<01:42, 42.78[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 56 |
+
{'loss': '5.475', 'grad_norm': '2.328', 'learning_rate': '0.0001557', 'epoch': '0.008763', 'train/total_time_seconds': '5.424', 'train/time_per_step_avg': '0.009508', 'train/epoch_time_elapsed': '27.47', 'train/estimated_remaining_minutes': '0.7789'}
|
| 57 |
+
{'loss': '5.397', 'grad_norm': '1.125', 'learning_rate': '0.0001617', 'epoch': '0.0091', 'train/total_time_seconds': '5.639', 'train/time_per_step_avg': '0.009766', 'train/epoch_time_elapsed': '27.99', 'train/estimated_remaining_minutes': '0.7762'}
|
| 58 |
+
{'loss': '5.331', 'grad_norm': '1.648', 'learning_rate': '0.0001677', 'epoch': '0.009437', 'train/total_time_seconds': '5.822', 'train/time_per_step_avg': '0.009752', 'train/epoch_time_elapsed': '28.43', 'train/estimated_remaining_minutes': '0.7693'}
|
| 59 |
+
{'loss': '5.257', 'grad_norm': '1.07', 'learning_rate': '0.0001737', 'epoch': '0.009774', 'train/total_time_seconds': '6.002', 'train/time_per_step_avg': '0.009698', 'train/epoch_time_elapsed': '28.86', 'train/estimated_remaining_minutes': '0.7623'}
|
| 60 |
+
{'loss': '5.152', 'grad_norm': '2.359', 'learning_rate': '0.0001797', 'epoch': '0.01011', 'train/total_time_seconds': '6.201', 'train/time_per_step_avg': '0.009613', 'train/epoch_time_elapsed': '29.32', 'train/estimated_remaining_minutes': '0.758'}
|
| 61 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 62 |
+
{'eval_loss': '5.118', 'eval_runtime': '7.457', 'eval_samples_per_second': '1278', 'eval_steps_per_second': '0.939', 'epoch': '0.01011', 'train/total_time_seconds': '6.201', 'train/time_per_step_avg': '0.009613', 'train/epoch_time_elapsed': '36.78', 'train/estimated_remaining_minutes': '0.758'}
|
| 63 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 64 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 65 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 363.80it/s]
|
| 66 |
+
14%|█▍ | 700/5000 [00:38<01:35, 45.02it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 67 |
+
{'loss': '5.097', 'grad_norm': '1.375', 'learning_rate': '0.0001857', 'epoch': '0.01045', 'train/total_time_seconds': '6.4', 'train/time_per_step_avg': '0.009761', 'train/epoch_time_elapsed': '37.25', 'train/estimated_remaining_minutes': '0.7536'}
|
| 68 |
+
{'loss': '5.02', 'grad_norm': '1.422', 'learning_rate': '0.0001917', 'epoch': '0.01079', 'train/total_time_seconds': '6.575', 'train/time_per_step_avg': '0.009358', 'train/epoch_time_elapsed': '37.68', 'train/estimated_remaining_minutes': '0.7465'}
|
| 69 |
+
{'loss': '4.98', 'grad_norm': '2.125', 'learning_rate': '0.0001977', 'epoch': '0.01112', 'train/total_time_seconds': '6.755', 'train/time_per_step_avg': '0.009331', 'train/epoch_time_elapsed': '38.12', 'train/estimated_remaining_minutes': '0.7403'}
|
| 70 |
+
{'loss': '4.946', 'grad_norm': '2.203', 'learning_rate': '0.0002037', 'epoch': '0.01146', 'train/total_time_seconds': '6.931', 'train/time_per_step_avg': '0.009292', 'train/epoch_time_elapsed': '38.55', 'train/estimated_remaining_minutes': '0.7339'}
|
| 71 |
+
{'loss': '4.866', 'grad_norm': '1.852', 'learning_rate': '0.0002097', 'epoch': '0.0118', 'train/total_time_seconds': '7.108', 'train/time_per_step_avg': '0.00907', 'train/epoch_time_elapsed': '38.98', 'train/estimated_remaining_minutes': '0.7278'}
|
| 72 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 73 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 74 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 75 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 361.67it/s]
|
| 76 |
+
16%|█▌ | 800/5000 [00:48<01:30, 46.66[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 77 |
+
{'loss': '4.842', 'grad_norm': '1.625', 'learning_rate': '0.0002157', 'epoch': '0.01213', 'train/total_time_seconds': '7.287', 'train/time_per_step_avg': '0.008866', 'train/epoch_time_elapsed': '39.44', 'train/estimated_remaining_minutes': '0.722'}
|
| 78 |
+
{'loss': '4.799', 'grad_norm': '3.406', 'learning_rate': '0.0002217', 'epoch': '0.01247', 'train/total_time_seconds': '7.461', 'train/time_per_step_avg': '0.008864', 'train/epoch_time_elapsed': '39.86', 'train/estimated_remaining_minutes': '0.7158'}
|
| 79 |
+
{'loss': '4.769', 'grad_norm': '2.219', 'learning_rate': '0.0002277', 'epoch': '0.01281', 'train/total_time_seconds': '7.635', 'train/time_per_step_avg': '0.008802', 'train/epoch_time_elapsed': '40.29', 'train/estimated_remaining_minutes': '0.7099'}
|
| 80 |
+
{'loss': '4.736', 'grad_norm': '1.641', 'learning_rate': '0.0002337', 'epoch': '0.01314', 'train/total_time_seconds': '7.813', 'train/time_per_step_avg': '0.008818', 'train/epoch_time_elapsed': '40.72', 'train/estimated_remaining_minutes': '0.7045'}
|
| 81 |
+
{'loss': '4.697', 'grad_norm': '1.688', 'learning_rate': '0.0002397', 'epoch': '0.01348', 'train/total_time_seconds': '7.987', 'train/time_per_step_avg': '0.008783', 'train/epoch_time_elapsed': '41.15', 'train/estimated_remaining_minutes': '0.6988'}
|
| 82 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 83 |
+
{'eval_loss': '4.677', 'eval_runtime': '7.439', 'eval_samples_per_second': '1281', 'eval_steps_per_second': '0.941', 'epoch': '0.01348', 'train/total_time_seconds': '7.987', 'train/time_per_step_avg': '0.008783', 'train/epoch_time_elapsed': '48.59', 'train/estimated_remaining_minutes': '0.6988'}
|
| 84 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 85 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 86 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 365.48it/s]
|
| 87 |
+
18%|█▊ | 900/5000 [00:50<01:31, 44.85it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 88 |
+
{'loss': '4.659', 'grad_norm': '1.648', 'learning_rate': '0.0002457', 'epoch': '0.01382', 'train/total_time_seconds': '8.165', 'train/time_per_step_avg': '0.008781', 'train/epoch_time_elapsed': '49.05', 'train/estimated_remaining_minutes': '0.6937'}
|
| 89 |
+
{'loss': '4.596', 'grad_norm': '3.234', 'learning_rate': '0.0002517', 'epoch': '0.01416', 'train/total_time_seconds': '8.341', 'train/time_per_step_avg': '0.008806', 'train/epoch_time_elapsed': '49.48', 'train/estimated_remaining_minutes': '0.6885'}
|
| 90 |
+
{'loss': '4.597', 'grad_norm': '1.578', 'learning_rate': '0.0002577', 'epoch': '0.01449', 'train/total_time_seconds': '8.521', 'train/time_per_step_avg': '0.008861', 'train/epoch_time_elapsed': '49.92', 'train/estimated_remaining_minutes': '0.6837'}
|
| 91 |
+
{'loss': '4.55', 'grad_norm': '1.32', 'learning_rate': '0.0002637', 'epoch': '0.01483', 'train/total_time_seconds': '8.697', 'train/time_per_step_avg': '0.008841', 'train/epoch_time_elapsed': '50.35', 'train/estimated_remaining_minutes': '0.6786'}
|
| 92 |
+
{'loss': '4.513', 'grad_norm': '2.281', 'learning_rate': '0.0002697', 'epoch': '0.01517', 'train/total_time_seconds': '8.875', 'train/time_per_step_avg': '0.008886', 'train/epoch_time_elapsed': '50.79', 'train/estimated_remaining_minutes': '0.6739'}
|
| 93 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 94 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 95 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 96 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 363.55it/s]
|
| 97 |
+
20%|██ | 1000/5000 [01:00<01:25, 46.5[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 98 |
+
{'loss': '4.476', 'grad_norm': '1.688', 'learning_rate': '0.0002757', 'epoch': '0.0155', 'train/total_time_seconds': '9.053', 'train/time_per_step_avg': '0.008877', 'train/epoch_time_elapsed': '51.24', 'train/estimated_remaining_minutes': '0.6691'}
|
| 99 |
+
{'loss': '4.433', 'grad_norm': '1.602', 'learning_rate': '0.0002817', 'epoch': '0.01584', 'train/total_time_seconds': '9.228', 'train/time_per_step_avg': '0.008866', 'train/epoch_time_elapsed': '51.67', 'train/estimated_remaining_minutes': '0.6643'}
|
| 100 |
+
{'loss': '4.402', 'grad_norm': '2.031', 'learning_rate': '0.0002877', 'epoch': '0.01618', 'train/total_time_seconds': '9.407', 'train/time_per_step_avg': '0.008858', 'train/epoch_time_elapsed': '52.1', 'train/estimated_remaining_minutes': '0.6598'}
|
| 101 |
+
{'loss': '4.383', 'grad_norm': '1.719', 'learning_rate': '0.0002937', 'epoch': '0.01651', 'train/total_time_seconds': '9.586', 'train/time_per_step_avg': '0.008889', 'train/epoch_time_elapsed': '52.53', 'train/estimated_remaining_minutes': '0.6553'}
|
| 102 |
+
{'loss': '4.368', 'grad_norm': '1.219', 'learning_rate': '0.0002997', 'epoch': '0.01685', 'train/total_time_seconds': '9.762', 'train/time_per_step_avg': '0.008866', 'train/epoch_time_elapsed': '52.96', 'train/estimated_remaining_minutes': '0.6508'}
|
| 103 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 104 |
+
{'eval_loss': '4.343', 'eval_runtime': '7.451', 'eval_samples_per_second': '1279', 'eval_steps_per_second': '0.939', 'epoch': '0.01685', 'train/total_time_seconds': '9.762', 'train/time_per_step_avg': '0.008866', 'train/epoch_time_elapsed': '60.42', 'train/estimated_remaining_minutes': '0.6508'}
|
| 105 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 106 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 107 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 360.00it/s]
|
| 108 |
+
22%|██▏ | 1100/5000 [01:02<01:25, 45.40it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 109 |
+
{'loss': '4.332', 'grad_norm': '1.641', 'learning_rate': '0.0003', 'epoch': '0.01719', 'train/total_time_seconds': '9.942', 'train/time_per_step_avg': '0.008889', 'train/epoch_time_elapsed': '60.88', 'train/estimated_remaining_minutes': '0.6465'}
|
| 110 |
+
{'loss': '4.315', 'grad_norm': '1.898', 'learning_rate': '0.0003', 'epoch': '0.01753', 'train/total_time_seconds': '10.12', 'train/time_per_step_avg': '0.008917', 'train/epoch_time_elapsed': '61.32', 'train/estimated_remaining_minutes': '0.6422'}
|
| 111 |
+
{'loss': '4.262', 'grad_norm': '1.883', 'learning_rate': '0.0003', 'epoch': '0.01786', 'train/total_time_seconds': '10.3', 'train/time_per_step_avg': '0.008911', 'train/epoch_time_elapsed': '61.75', 'train/estimated_remaining_minutes': '0.638'}
|
| 112 |
+
{'loss': '4.262', 'grad_norm': '2.312', 'learning_rate': '0.0003', 'epoch': '0.0182', 'train/total_time_seconds': '10.48', 'train/time_per_step_avg': '0.008912', 'train/epoch_time_elapsed': '62.18', 'train/estimated_remaining_minutes': '0.6338'}
|
| 113 |
+
{'loss': '4.222', 'grad_norm': '2.25', 'learning_rate': '0.0003', 'epoch': '0.01854', 'train/total_time_seconds': '10.65', 'train/time_per_step_avg': '0.00891', 'train/epoch_time_elapsed': '62.61', 'train/estimated_remaining_minutes': '0.6295'}
|
| 114 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 115 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 116 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 117 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 362.99it/s]
|
| 118 |
+
24%|██▍ | 1200/5000 [01:12<01:21, 46.5[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 119 |
+
{'loss': '4.216', 'grad_norm': '1.633', 'learning_rate': '0.0003', 'epoch': '0.01887', 'train/total_time_seconds': '10.85', 'train/time_per_step_avg': '0.009061', 'train/epoch_time_elapsed': '63.12', 'train/estimated_remaining_minutes': '0.6263'}
|
| 120 |
+
{'loss': '4.189', 'grad_norm': '1.844', 'learning_rate': '0.0003', 'epoch': '0.01921', 'train/total_time_seconds': '11.03', 'train/time_per_step_avg': '0.009092', 'train/epoch_time_elapsed': '63.55', 'train/estimated_remaining_minutes': '0.6224'}
|
| 121 |
+
{'loss': '4.164', 'grad_norm': '2.547', 'learning_rate': '0.0003', 'epoch': '0.01955', 'train/total_time_seconds': '11.21', 'train/time_per_step_avg': '0.009093', 'train/epoch_time_elapsed': '63.99', 'train/estimated_remaining_minutes': '0.6183'}
|
| 122 |
+
{'loss': '4.162', 'grad_norm': '1.461', 'learning_rate': '0.0003', 'epoch': '0.01989', 'train/total_time_seconds': '11.38', 'train/time_per_step_avg': '0.009057', 'train/epoch_time_elapsed': '64.42', 'train/estimated_remaining_minutes': '0.6141'}
|
| 123 |
+
{'loss': '4.131', 'grad_norm': '1.562', 'learning_rate': '0.0003', 'epoch': '0.02022', 'train/total_time_seconds': '11.56', 'train/time_per_step_avg': '0.009057', 'train/epoch_time_elapsed': '64.85', 'train/estimated_remaining_minutes': '0.61'}
|
| 124 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 125 |
+
{'eval_loss': '4.134', 'eval_runtime': '7.466', 'eval_samples_per_second': '1276', 'eval_steps_per_second': '0.938', 'epoch': '0.02022', 'train/total_time_seconds': '11.56', 'train/time_per_step_avg': '0.009057', 'train/epoch_time_elapsed': '72.32', 'train/estimated_remaining_minutes': '0.61'}
|
| 126 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 127 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 128 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 343.71it/s]
|
| 129 |
+
26%|██▌ | 1300/5000 [01:14<01:21, 45.35it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 130 |
+
{'loss': '4.11', 'grad_norm': '1.289', 'learning_rate': '0.0003', 'epoch': '0.02056', 'train/total_time_seconds': '11.74', 'train/time_per_step_avg': '0.00896', 'train/epoch_time_elapsed': '72.8', 'train/estimated_remaining_minutes': '0.6064'}
|
| 131 |
+
{'loss': '4.086', 'grad_norm': '2.109', 'learning_rate': '0.0003', 'epoch': '0.0209', 'train/total_time_seconds': '11.93', 'train/time_per_step_avg': '0.009023', 'train/epoch_time_elapsed': '73.24', 'train/estimated_remaining_minutes': '0.603'}
|
| 132 |
+
{'loss': '4.069', 'grad_norm': '1.586', 'learning_rate': '0.0003', 'epoch': '0.02123', 'train/total_time_seconds': '12.11', 'train/time_per_step_avg': '0.009003', 'train/epoch_time_elapsed': '73.68', 'train/estimated_remaining_minutes': '0.599'}
|
| 133 |
+
{'loss': '4.091', 'grad_norm': '1.789', 'learning_rate': '0.0003', 'epoch': '0.02157', 'train/total_time_seconds': '12.28', 'train/time_per_step_avg': '0.009016', 'train/epoch_time_elapsed': '74.11', 'train/estimated_remaining_minutes': '0.595'}
|
| 134 |
+
{'loss': '4.062', 'grad_norm': '1.484', 'learning_rate': '0.0003', 'epoch': '0.02191', 'train/total_time_seconds': '12.46', 'train/time_per_step_avg': '0.00901', 'train/epoch_time_elapsed': '74.54', 'train/estimated_remaining_minutes': '0.591'}
|
| 135 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 136 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 137 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 138 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 356.33it/s]
|
| 139 |
+
28%|██▊ | 1400/5000 [01:24<01:17, 46.6[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 140 |
+
{'loss': '4.038', 'grad_norm': '1.562', 'learning_rate': '0.0003', 'epoch': '0.02224', 'train/total_time_seconds': '12.64', 'train/time_per_step_avg': '0.008917', 'train/epoch_time_elapsed': '74.99', 'train/estimated_remaining_minutes': '0.5871'}
|
| 141 |
+
{'loss': '4.031', 'grad_norm': '1.359', 'learning_rate': '0.0003', 'epoch': '0.02258', 'train/total_time_seconds': '12.81', 'train/time_per_step_avg': '0.008809', 'train/epoch_time_elapsed': '75.42', 'train/estimated_remaining_minutes': '0.5832'}
|
| 142 |
+
{'loss': '4.002', 'grad_norm': '1.438', 'learning_rate': '0.0003', 'epoch': '0.02292', 'train/total_time_seconds': '12.99', 'train/time_per_step_avg': '0.008808', 'train/epoch_time_elapsed': '75.85', 'train/estimated_remaining_minutes': '0.5794'}
|
| 143 |
+
{'loss': '4.032', 'grad_norm': '1.773', 'learning_rate': '0.0003', 'epoch': '0.02326', 'train/total_time_seconds': '13.16', 'train/time_per_step_avg': '0.008808', 'train/epoch_time_elapsed': '76.28', 'train/estimated_remaining_minutes': '0.5756'}
|
| 144 |
+
{'loss': '4.004', 'grad_norm': '1.461', 'learning_rate': '0.0003', 'epoch': '0.02359', 'train/total_time_seconds': '13.34', 'train/time_per_step_avg': '0.008812', 'train/epoch_time_elapsed': '76.71', 'train/estimated_remaining_minutes': '0.5718'}
|
| 145 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 146 |
+
{'eval_loss': '4', 'eval_runtime': '7.501', 'eval_samples_per_second': '1270', 'eval_steps_per_second': '0.933', 'epoch': '0.02359', 'train/total_time_seconds': '13.34', 'train/time_per_step_avg': '0.008812', 'train/epoch_time_elapsed': '84.21', 'train/estimated_remaining_minutes': '0.5718'}
|
| 147 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 148 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 149 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 369.12it/s]
|
| 150 |
+
30%|███ | 1500/5000 [01:26<01:16, 45.66it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 151 |
+
{'loss': '4.006', 'grad_norm': '1.531', 'learning_rate': '0.0003', 'epoch': '0.02393', 'train/total_time_seconds': '13.52', 'train/time_per_step_avg': '0.008825', 'train/epoch_time_elapsed': '84.67', 'train/estimated_remaining_minutes': '0.568'}
|
| 152 |
+
{'loss': '3.973', 'grad_norm': '1.359', 'learning_rate': '0.0003', 'epoch': '0.02427', 'train/total_time_seconds': '13.69', 'train/time_per_step_avg': '0.008785', 'train/epoch_time_elapsed': '85.09', 'train/estimated_remaining_minutes': '0.5641'}
|
| 153 |
+
{'loss': '4.007', 'grad_norm': '1.375', 'learning_rate': '0.0003', 'epoch': '0.0246', 'train/total_time_seconds': '13.87', 'train/time_per_step_avg': '0.008768', 'train/epoch_time_elapsed': '85.52', 'train/estimated_remaining_minutes': '0.5603'}
|
| 154 |
+
{'loss': '3.955', 'grad_norm': '2.031', 'learning_rate': '0.0003', 'epoch': '0.02494', 'train/total_time_seconds': '14.04', 'train/time_per_step_avg': '0.008757', 'train/epoch_time_elapsed': '85.95', 'train/estimated_remaining_minutes': '0.5566'}
|
| 155 |
+
{'loss': '3.959', 'grad_norm': '1.359', 'learning_rate': '0.0003', 'epoch': '0.02528', 'train/total_time_seconds': '14.22', 'train/time_per_step_avg': '0.008745', 'train/epoch_time_elapsed': '86.38', 'train/estimated_remaining_minutes': '0.5528'}
|
| 156 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 157 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 158 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 159 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 366.83it/s]
|
| 160 |
+
32%|███▏ | 1600/5000 [01:36<01:13, 46.0[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 161 |
+
{'loss': '3.925', 'grad_norm': '1.648', 'learning_rate': '0.0003', 'epoch': '0.02562', 'train/total_time_seconds': '14.39', 'train/time_per_step_avg': '0.008722', 'train/epoch_time_elapsed': '86.84', 'train/estimated_remaining_minutes': '0.5491'}
|
| 162 |
+
{'loss': '3.916', 'grad_norm': '1.531', 'learning_rate': '0.0003', 'epoch': '0.02595', 'train/total_time_seconds': '14.56', 'train/time_per_step_avg': '0.008724', 'train/epoch_time_elapsed': '87.27', 'train/estimated_remaining_minutes': '0.5453'}
|
| 163 |
+
{'loss': '3.9', 'grad_norm': '1.469', 'learning_rate': '0.0003', 'epoch': '0.02629', 'train/total_time_seconds': '14.74', 'train/time_per_step_avg': '0.008734', 'train/epoch_time_elapsed': '87.69', 'train/estimated_remaining_minutes': '0.5417'}
|
| 164 |
+
{'loss': '3.896', 'grad_norm': '1.508', 'learning_rate': '0.0003', 'epoch': '0.02663', 'train/total_time_seconds': '14.91', 'train/time_per_step_avg': '0.00871', 'train/epoch_time_elapsed': '88.12', 'train/estimated_remaining_minutes': '0.5379'}
|
| 165 |
+
{'loss': '3.939', 'grad_norm': '1.445', 'learning_rate': '0.0003', 'epoch': '0.02696', 'train/total_time_seconds': '15.1', 'train/time_per_step_avg': '0.008816', 'train/epoch_time_elapsed': '88.57', 'train/estimated_remaining_minutes': '0.5347'}
|
| 166 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 167 |
+
{'eval_loss': '3.91', 'eval_runtime': '7.782', 'eval_samples_per_second': '1224', 'eval_steps_per_second': '0.899', 'epoch': '0.02696', 'train/total_time_seconds': '15.1', 'train/time_per_step_avg': '0.008816', 'train/epoch_time_elapsed': '96.35', 'train/estimated_remaining_minutes': '0.5347'}
|
| 168 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 169 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 170 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 362.39it/s]
|
| 171 |
+
34%|███▍ | 1700/5000 [01:38<01:12, 45.83it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 172 |
+
{'loss': '3.894', 'grad_norm': '1.523', 'learning_rate': '0.0003', 'epoch': '0.0273', 'train/total_time_seconds': '15.27', 'train/time_per_step_avg': '0.008808', 'train/epoch_time_elapsed': '96.8', 'train/estimated_remaining_minutes': '0.531'}
|
| 173 |
+
{'loss': '3.874', 'grad_norm': '1.656', 'learning_rate': '0.0003', 'epoch': '0.02764', 'train/total_time_seconds': '15.44', 'train/time_per_step_avg': '0.008805', 'train/epoch_time_elapsed': '97.23', 'train/estimated_remaining_minutes': '0.5273'}
|
| 174 |
+
{'loss': '3.875', 'grad_norm': '1.805', 'learning_rate': '0.0003', 'epoch': '0.02797', 'train/total_time_seconds': '15.62', 'train/time_per_step_avg': '0.008784', 'train/epoch_time_elapsed': '97.66', 'train/estimated_remaining_minutes': '0.5237'}
|
| 175 |
+
{'loss': '3.885', 'grad_norm': '1.516', 'learning_rate': '0.0003', 'epoch': '0.02831', 'train/total_time_seconds': '15.79', 'train/time_per_step_avg': '0.008792', 'train/epoch_time_elapsed': '98.08', 'train/estimated_remaining_minutes': '0.5201'}
|
| 176 |
+
{'loss': '3.901', 'grad_norm': '1.641', 'learning_rate': '0.0003', 'epoch': '0.02865', 'train/total_time_seconds': '15.96', 'train/time_per_step_avg': '0.008662', 'train/epoch_time_elapsed': '98.51', 'train/estimated_remaining_minutes': '0.5165'}
|
| 177 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 178 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 179 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 180 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 362.11it/s]
|
| 181 |
+
36%|███▌ | 1800/5000 [01:48<01:10, 45.7[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 182 |
+
{'loss': '3.914', 'grad_norm': '1.359', 'learning_rate': '0.0003', 'epoch': '0.02899', 'train/total_time_seconds': '16.14', 'train/time_per_step_avg': '0.008648', 'train/epoch_time_elapsed': '98.96', 'train/estimated_remaining_minutes': '0.5128'}
|
| 183 |
+
{'loss': '3.865', 'grad_norm': '1.492', 'learning_rate': '0.0003', 'epoch': '0.02932', 'train/total_time_seconds': '16.32', 'train/time_per_step_avg': '0.008794', 'train/epoch_time_elapsed': '99.41', 'train/estimated_remaining_minutes': '0.5097'}
|
| 184 |
+
{'loss': '3.865', 'grad_norm': '1.922', 'learning_rate': '0.0003', 'epoch': '0.02966', 'train/total_time_seconds': '16.5', 'train/time_per_step_avg': '0.008799', 'train/epoch_time_elapsed': '99.84', 'train/estimated_remaining_minutes': '0.5062'}
|
| 185 |
+
{'loss': '3.826', 'grad_norm': '1.781', 'learning_rate': '0.0003', 'epoch': '0.03', 'train/total_time_seconds': '16.67', 'train/time_per_step_avg': '0.008814', 'train/epoch_time_elapsed': '100.3', 'train/estimated_remaining_minutes': '0.5027'}
|
| 186 |
+
{'loss': '3.843', 'grad_norm': '1.398', 'learning_rate': '0.0003', 'epoch': '0.03033', 'train/total_time_seconds': '16.85', 'train/time_per_step_avg': '0.008869', 'train/epoch_time_elapsed': '100.7', 'train/estimated_remaining_minutes': '0.4993'}
|
| 187 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 188 |
+
{'eval_loss': '3.846', 'eval_runtime': '7.438', 'eval_samples_per_second': '1281', 'eval_steps_per_second': '0.941', 'epoch': '0.03033', 'train/total_time_seconds': '16.85', 'train/time_per_step_avg': '0.008869', 'train/epoch_time_elapsed': '108.2', 'train/estimated_remaining_minutes': '0.4993'}
|
| 189 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 190 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 191 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 369.54it/s]
|
| 192 |
+
38%|███▊ | 1900/5000 [01:50<01:10, 43.88it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 193 |
+
{'loss': '3.854', 'grad_norm': '1.289', 'learning_rate': '0.0003', 'epoch': '0.03067', 'train/total_time_seconds': '17.03', 'train/time_per_step_avg': '0.008931', 'train/epoch_time_elapsed': '108.6', 'train/estimated_remaining_minutes': '0.4959'}
|
| 194 |
+
{'loss': '3.843', 'grad_norm': '1.531', 'learning_rate': '0.0003', 'epoch': '0.03101', 'train/total_time_seconds': '17.2', 'train/time_per_step_avg': '0.008786', 'train/epoch_time_elapsed': '109', 'train/estimated_remaining_minutes': '0.4924'}
|
| 195 |
+
{'loss': '3.841', 'grad_norm': '1.641', 'learning_rate': '0.0003', 'epoch': '0.03134', 'train/total_time_seconds': '17.37', 'train/time_per_step_avg': '0.008777', 'train/epoch_time_elapsed': '109.5', 'train/estimated_remaining_minutes': '0.4889'}
|
| 196 |
+
{'loss': '3.831', 'grad_norm': '1.109', 'learning_rate': '0.0003', 'epoch': '0.03168', 'train/total_time_seconds': '17.56', 'train/time_per_step_avg': '0.008864', 'train/epoch_time_elapsed': '109.9', 'train/estimated_remaining_minutes': '0.4857'}
|
| 197 |
+
{'loss': '3.804', 'grad_norm': '1.414', 'learning_rate': '0.0003', 'epoch': '0.03202', 'train/total_time_seconds': '17.74', 'train/time_per_step_avg': '0.008891', 'train/epoch_time_elapsed': '110.4', 'train/estimated_remaining_minutes': '0.4824'}
|
| 198 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 199 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 200 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 201 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 346.09it/s]
|
| 202 |
+
40%|████ | 2000/5000 [02:00<01:04, 46.8[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 203 |
+
{'loss': '3.816', 'grad_norm': '1.836', 'learning_rate': '0.0003', 'epoch': '0.03236', 'train/total_time_seconds': '17.91', 'train/time_per_step_avg': '0.008847', 'train/epoch_time_elapsed': '110.8', 'train/estimated_remaining_minutes': '0.4789'}
|
| 204 |
+
{'loss': '3.799', 'grad_norm': '1.5', 'learning_rate': '0.0003', 'epoch': '0.03269', 'train/total_time_seconds': '18.09', 'train/time_per_step_avg': '0.008862', 'train/epoch_time_elapsed': '111.2', 'train/estimated_remaining_minutes': '0.4755'}
|
| 205 |
+
{'loss': '3.801', 'grad_norm': '2.078', 'learning_rate': '0.0003', 'epoch': '0.03303', 'train/total_time_seconds': '18.26', 'train/time_per_step_avg': '0.008862', 'train/epoch_time_elapsed': '111.7', 'train/estimated_remaining_minutes': '0.4721'}
|
| 206 |
+
{'loss': '3.79', 'grad_norm': '1.641', 'learning_rate': '0.0003', 'epoch': '0.03337', 'train/total_time_seconds': '18.44', 'train/time_per_step_avg': '0.00879', 'train/epoch_time_elapsed': '112.1', 'train/estimated_remaining_minutes': '0.4687'}
|
| 207 |
+
{'loss': '3.81', 'grad_norm': '1.695', 'learning_rate': '0.0003', 'epoch': '0.0337', 'train/total_time_seconds': '18.61', 'train/time_per_step_avg': '0.008721', 'train/epoch_time_elapsed': '112.5', 'train/estimated_remaining_minutes': '0.4653'}
|
| 208 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 209 |
+
{'eval_loss': '3.798', 'eval_runtime': '7.486', 'eval_samples_per_second': '1273', 'eval_steps_per_second': '0.935', 'epoch': '0.0337', 'train/total_time_seconds': '18.61', 'train/time_per_step_avg': '0.008721', 'train/epoch_time_elapsed': '120', 'train/estimated_remaining_minutes': '0.4653'}
|
| 210 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 211 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 212 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 366.28it/s]
|
| 213 |
+
42%|████▏ | 2100/5000 [02:02<01:03, 45.65it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 214 |
+
{'loss': '3.801', 'grad_norm': '1.773', 'learning_rate': '0.0003', 'epoch': '0.03404', 'train/total_time_seconds': '18.79', 'train/time_per_step_avg': '0.008809', 'train/epoch_time_elapsed': '120.5', 'train/estimated_remaining_minutes': '0.4621'}
|
| 215 |
+
{'loss': '3.787', 'grad_norm': '1.523', 'learning_rate': '0.0003', 'epoch': '0.03438', 'train/total_time_seconds': '18.97', 'train/time_per_step_avg': '0.008819', 'train/epoch_time_elapsed': '120.9', 'train/estimated_remaining_minutes': '0.4587'}
|
| 216 |
+
{'loss': '3.797', 'grad_norm': '1.469', 'learning_rate': '0.0003', 'epoch': '0.03472', 'train/total_time_seconds': '19.14', 'train/time_per_step_avg': '0.008838', 'train/epoch_time_elapsed': '121.3', 'train/estimated_remaining_minutes': '0.4554'}
|
| 217 |
+
{'loss': '3.76', 'grad_norm': '1.734', 'learning_rate': '0.0003', 'epoch': '0.03505', 'train/total_time_seconds': '19.33', 'train/time_per_step_avg': '0.00888', 'train/epoch_time_elapsed': '121.8', 'train/estimated_remaining_minutes': '0.4522'}
|
| 218 |
+
{'loss': '3.766', 'grad_norm': '1.477', 'learning_rate': '0.0003', 'epoch': '0.03539', 'train/total_time_seconds': '19.5', 'train/time_per_step_avg': '0.008889', 'train/epoch_time_elapsed': '122.2', 'train/estimated_remaining_minutes': '0.4488'}
|
| 219 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 220 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 221 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 222 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 367.02it/s]
|
| 223 |
+
44%|████▍ | 2200/5000 [02:11<00:59, 47.0[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 224 |
+
{'loss': '3.766', 'grad_norm': '1.5', 'learning_rate': '0.0003', 'epoch': '0.03573', 'train/total_time_seconds': '19.68', 'train/time_per_step_avg': '0.00883', 'train/epoch_time_elapsed': '122.7', 'train/estimated_remaining_minutes': '0.4455'}
|
| 225 |
+
{'loss': '3.771', 'grad_norm': '1.508', 'learning_rate': '0.0003', 'epoch': '0.03606', 'train/total_time_seconds': '19.86', 'train/time_per_step_avg': '0.008901', 'train/epoch_time_elapsed': '123.1', 'train/estimated_remaining_minutes': '0.4424'}
|
| 226 |
+
{'loss': '3.744', 'grad_norm': '1.945', 'learning_rate': '0.0003', 'epoch': '0.0364', 'train/total_time_seconds': '20.03', 'train/time_per_step_avg': '0.0089', 'train/epoch_time_elapsed': '123.5', 'train/estimated_remaining_minutes': '0.439'}
|
| 227 |
+
{'loss': '3.753', 'grad_norm': '1.391', 'learning_rate': '0.0003', 'epoch': '0.03674', 'train/total_time_seconds': '20.21', 'train/time_per_step_avg': '0.008871', 'train/epoch_time_elapsed': '124', 'train/estimated_remaining_minutes': '0.4358'}
|
| 228 |
+
{'loss': '3.753', 'grad_norm': '1.516', 'learning_rate': '0.0003', 'epoch': '0.03707', 'train/total_time_seconds': '20.39', 'train/time_per_step_avg': '0.008849', 'train/epoch_time_elapsed': '124.4', 'train/estimated_remaining_minutes': '0.4324'}
|
| 229 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 230 |
+
{'eval_loss': '3.762', 'eval_runtime': '7.429', 'eval_samples_per_second': '1282', 'eval_steps_per_second': '0.942', 'epoch': '0.03707', 'train/total_time_seconds': '20.39', 'train/time_per_step_avg': '0.008849', 'train/epoch_time_elapsed': '131.8', 'train/estimated_remaining_minutes': '0.4324'}
|
| 231 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 232 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 233 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 367.24it/s]
|
| 234 |
+
46%|████▌ | 2300/5000 [02:13<00:58, 45.99it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 235 |
+
{'loss': '3.753', 'grad_norm': '1.766', 'learning_rate': '0.0003', 'epoch': '0.03741', 'train/total_time_seconds': '20.56', 'train/time_per_step_avg': '0.00884', 'train/epoch_time_elapsed': '132.3', 'train/estimated_remaining_minutes': '0.4291'}
|
| 236 |
+
{'loss': '3.759', 'grad_norm': '1.766', 'learning_rate': '0.0003', 'epoch': '0.03775', 'train/total_time_seconds': '20.74', 'train/time_per_step_avg': '0.008766', 'train/epoch_time_elapsed': '132.7', 'train/estimated_remaining_minutes': '0.4258'}
|
| 237 |
+
{'loss': '3.744', 'grad_norm': '1.719', 'learning_rate': '0.0003', 'epoch': '0.03809', 'train/total_time_seconds': '20.92', 'train/time_per_step_avg': '0.008812', 'train/epoch_time_elapsed': '133.1', 'train/estimated_remaining_minutes': '0.4226'}
|
| 238 |
+
{'loss': '3.729', 'grad_norm': '1.859', 'learning_rate': '0.0003', 'epoch': '0.03842', 'train/total_time_seconds': '21.09', 'train/time_per_step_avg': '0.008765', 'train/epoch_time_elapsed': '133.6', 'train/estimated_remaining_minutes': '0.4193'}
|
| 239 |
+
{'loss': '3.75', 'grad_norm': '1.609', 'learning_rate': '0.0003', 'epoch': '0.03876', 'train/total_time_seconds': '21.26', 'train/time_per_step_avg': '0.008778', 'train/epoch_time_elapsed': '134', 'train/estimated_remaining_minutes': '0.416'}
|
| 240 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 241 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 242 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 243 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 357.51it/s]
|
| 244 |
+
48%|████▊ | 2400/5000 [02:23<00:55, 46.7[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 245 |
+
{'loss': '3.717', 'grad_norm': '1.516', 'learning_rate': '0.0003', 'epoch': '0.0391', 'train/total_time_seconds': '21.44', 'train/time_per_step_avg': '0.008763', 'train/epoch_time_elapsed': '134.4', 'train/estimated_remaining_minutes': '0.4127'}
|
| 246 |
+
{'loss': '3.749', 'grad_norm': '1.344', 'learning_rate': '0.0003', 'epoch': '0.03943', 'train/total_time_seconds': '21.61', 'train/time_per_step_avg': '0.008761', 'train/epoch_time_elapsed': '134.9', 'train/estimated_remaining_minutes': '0.4095'}
|
| 247 |
+
{'loss': '3.749', 'grad_norm': '1.586', 'learning_rate': '0.0003', 'epoch': '0.03977', 'train/total_time_seconds': '21.79', 'train/time_per_step_avg': '0.008712', 'train/epoch_time_elapsed': '135.3', 'train/estimated_remaining_minutes': '0.4062'}
|
| 248 |
+
{'loss': '3.721', 'grad_norm': '1.422', 'learning_rate': '0.0003', 'epoch': '0.04011', 'train/total_time_seconds': '21.96', 'train/time_per_step_avg': '0.008734', 'train/epoch_time_elapsed': '135.7', 'train/estimated_remaining_minutes': '0.403'}
|
| 249 |
+
{'loss': '3.724', 'grad_norm': '1.711', 'learning_rate': '0.0003', 'epoch': '0.04044', 'train/total_time_seconds': '22.14', 'train/time_per_step_avg': '0.00876', 'train/epoch_time_elapsed': '136.1', 'train/estimated_remaining_minutes': '0.3997'}
|
| 250 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 251 |
+
{'eval_loss': '3.733', 'eval_runtime': '7.418', 'eval_samples_per_second': '1284', 'eval_steps_per_second': '0.944', 'epoch': '0.04044', 'train/total_time_seconds': '22.14', 'train/time_per_step_avg': '0.00876', 'train/epoch_time_elapsed': '143.6', 'train/estimated_remaining_minutes': '0.3997'}
|
| 252 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 253 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 254 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 225.17it/s]
|
| 255 |
+
50%|█████ | 2500/5000 [02:25<00:54, 45.48it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 256 |
+
{'loss': '3.712', 'grad_norm': '1.547', 'learning_rate': '0.0003', 'epoch': '0.04078', 'train/total_time_seconds': '22.32', 'train/time_per_step_avg': '0.008778', 'train/epoch_time_elapsed': '144', 'train/estimated_remaining_minutes': '0.3965'}
|
| 257 |
+
{'loss': '3.696', 'grad_norm': '1.547', 'learning_rate': '0.0003', 'epoch': '0.04112', 'train/total_time_seconds': '22.49', 'train/time_per_step_avg': '0.008795', 'train/epoch_time_elapsed': '144.4', 'train/estimated_remaining_minutes': '0.3933'}
|
| 258 |
+
{'loss': '3.701', 'grad_norm': '1.508', 'learning_rate': '0.0003', 'epoch': '0.04146', 'train/total_time_seconds': '22.67', 'train/time_per_step_avg': '0.008795', 'train/epoch_time_elapsed': '144.9', 'train/estimated_remaining_minutes': '0.3901'}
|
| 259 |
+
{'loss': '3.731', 'grad_norm': '1.328', 'learning_rate': '0.0003', 'epoch': '0.04179', 'train/total_time_seconds': '22.84', 'train/time_per_step_avg': '0.008761', 'train/epoch_time_elapsed': '145.3', 'train/estimated_remaining_minutes': '0.3868'}
|
| 260 |
+
{'loss': '3.728', 'grad_norm': '1.742', 'learning_rate': '0.0003', 'epoch': '0.04213', 'train/total_time_seconds': '23.01', 'train/time_per_step_avg': '0.008735', 'train/epoch_time_elapsed': '145.7', 'train/estimated_remaining_minutes': '0.3835'}
|
| 261 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 262 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 263 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 264 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 365.10it/s]
|
| 265 |
+
52%|█████▏ | 2600/5000 [02:35<00:50, 47.2[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 266 |
+
{'loss': '3.724', 'grad_norm': '1.312', 'learning_rate': '0.0003', 'epoch': '0.04247', 'train/total_time_seconds': '23.19', 'train/time_per_step_avg': '0.008701', 'train/epoch_time_elapsed': '146.2', 'train/estimated_remaining_minutes': '0.3803'}
|
| 267 |
+
{'loss': '3.72', 'grad_norm': '1.727', 'learning_rate': '0.0003', 'epoch': '0.0428', 'train/total_time_seconds': '23.36', 'train/time_per_step_avg': '0.008655', 'train/epoch_time_elapsed': '146.6', 'train/estimated_remaining_minutes': '0.377'}
|
| 268 |
+
{'loss': '3.702', 'grad_norm': '2.094', 'learning_rate': '0.0003', 'epoch': '0.04314', 'train/total_time_seconds': '23.53', 'train/time_per_step_avg': '0.008653', 'train/epoch_time_elapsed': '147', 'train/estimated_remaining_minutes': '0.3738'}
|
| 269 |
+
{'loss': '3.713', 'grad_norm': '1.789', 'learning_rate': '0.0003', 'epoch': '0.04348', 'train/total_time_seconds': '23.7', 'train/time_per_step_avg': '0.008646', 'train/epoch_time_elapsed': '147.4', 'train/estimated_remaining_minutes': '0.3706'}
|
| 270 |
+
{'loss': '3.695', 'grad_norm': '1.641', 'learning_rate': '0.0003', 'epoch': '0.04382', 'train/total_time_seconds': '23.88', 'train/time_per_step_avg': '0.00863', 'train/epoch_time_elapsed': '147.9', 'train/estimated_remaining_minutes': '0.3673'}
|
| 271 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 272 |
+
{'eval_loss': '3.698', 'eval_runtime': '7.608', 'eval_samples_per_second': '1252', 'eval_steps_per_second': '0.92', 'epoch': '0.04382', 'train/total_time_seconds': '23.88', 'train/time_per_step_avg': '0.00863', 'train/epoch_time_elapsed': '155.5', 'train/estimated_remaining_minutes': '0.3673'}
|
| 273 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 274 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 275 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 352.67it/s]
|
| 276 |
+
54%|█████▍ | 2700/5000 [02:37<00:50, 45.39it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 277 |
+
{'loss': '3.661', 'grad_norm': '1.289', 'learning_rate': '0.0003', 'epoch': '0.04415', 'train/total_time_seconds': '24.05', 'train/time_per_step_avg': '0.008667', 'train/epoch_time_elapsed': '155.9', 'train/estimated_remaining_minutes': '0.3642'}
|
| 278 |
+
{'loss': '3.69', 'grad_norm': '1.703', 'learning_rate': '0.0003', 'epoch': '0.04449', 'train/total_time_seconds': '24.23', 'train/time_per_step_avg': '0.008721', 'train/epoch_time_elapsed': '156.4', 'train/estimated_remaining_minutes': '0.361'}
|
| 279 |
+
{'loss': '3.728', 'grad_norm': '1.523', 'learning_rate': '0.0003', 'epoch': '0.04483', 'train/total_time_seconds': '24.42', 'train/time_per_step_avg': '0.008879', 'train/epoch_time_elapsed': '156.8', 'train/estimated_remaining_minutes': '0.358'}
|
| 280 |
+
{'loss': '3.673', 'grad_norm': '1.594', 'learning_rate': '0.0003', 'epoch': '0.04516', 'train/total_time_seconds': '24.59', 'train/time_per_step_avg': '0.008906', 'train/epoch_time_elapsed': '157.3', 'train/estimated_remaining_minutes': '0.3548'}
|
| 281 |
+
{'loss': '3.679', 'grad_norm': '1.516', 'learning_rate': '0.0003', 'epoch': '0.0455', 'train/total_time_seconds': '24.77', 'train/time_per_step_avg': '0.008927', 'train/epoch_time_elapsed': '157.7', 'train/estimated_remaining_minutes': '0.3516'}
|
| 282 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 283 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 284 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 285 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 364.63it/s]
|
| 286 |
+
56%|█████▌ | 2800/5000 [02:47<00:47, 46.8[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 287 |
+
{'loss': '3.689', 'grad_norm': '1.492', 'learning_rate': '0.0003', 'epoch': '0.04584', 'train/total_time_seconds': '24.95', 'train/time_per_step_avg': '0.008949', 'train/epoch_time_elapsed': '158.1', 'train/estimated_remaining_minutes': '0.3485'}
|
| 288 |
+
{'loss': '3.657', 'grad_norm': '1.383', 'learning_rate': '0.0003', 'epoch': '0.04617', 'train/total_time_seconds': '25.13', 'train/time_per_step_avg': '0.009017', 'train/epoch_time_elapsed': '158.6', 'train/estimated_remaining_minutes': '0.3455'}
|
| 289 |
+
{'loss': '3.708', 'grad_norm': '1.812', 'learning_rate': '0.0003', 'epoch': '0.04651', 'train/total_time_seconds': '25.31', 'train/time_per_step_avg': '0.008923', 'train/epoch_time_elapsed': '159', 'train/estimated_remaining_minutes': '0.3424'}
|
| 290 |
+
{'loss': '3.67', 'grad_norm': '1.828', 'learning_rate': '0.0003', 'epoch': '0.04685', 'train/total_time_seconds': '25.49', 'train/time_per_step_avg': '0.00893', 'train/epoch_time_elapsed': '159.4', 'train/estimated_remaining_minutes': '0.3392'}
|
| 291 |
+
{'loss': '3.652', 'grad_norm': '1.477', 'learning_rate': '0.0003', 'epoch': '0.04719', 'train/total_time_seconds': '25.66', 'train/time_per_step_avg': '0.008942', 'train/epoch_time_elapsed': '159.9', 'train/estimated_remaining_minutes': '0.3361'}
|
| 292 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 293 |
+
{'eval_loss': '3.671', 'eval_runtime': '7.417', 'eval_samples_per_second': '1285', 'eval_steps_per_second': '0.944', 'epoch': '0.04719', 'train/total_time_seconds': '25.66', 'train/time_per_step_avg': '0.008942', 'train/epoch_time_elapsed': '167.3', 'train/estimated_remaining_minutes': '0.3361'}
|
| 294 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 295 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 296 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 193.85it/s]
|
| 297 |
+
58%|█████▊ | 2900/5000 [02:49<00:45, 45.75it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 298 |
+
{'loss': '3.684', 'grad_norm': '1.172', 'learning_rate': '0.0003', 'epoch': '0.04752', 'train/total_time_seconds': '25.84', 'train/time_per_step_avg': '0.008939', 'train/epoch_time_elapsed': '167.7', 'train/estimated_remaining_minutes': '0.3329'}
|
| 299 |
+
{'loss': '3.651', 'grad_norm': '1.672', 'learning_rate': '0.0003', 'epoch': '0.04786', 'train/total_time_seconds': '26.02', 'train/time_per_step_avg': '0.008859', 'train/epoch_time_elapsed': '168.2', 'train/estimated_remaining_minutes': '0.3298'}
|
| 300 |
+
{'loss': '3.654', 'grad_norm': '1.898', 'learning_rate': '0.0003', 'epoch': '0.0482', 'train/total_time_seconds': '26.19', 'train/time_per_step_avg': '0.008817', 'train/epoch_time_elapsed': '168.6', 'train/estimated_remaining_minutes': '0.3267'}
|
| 301 |
+
{'loss': '3.658', 'grad_norm': '1.469', 'learning_rate': '0.0003', 'epoch': '0.04853', 'train/total_time_seconds': '26.37', 'train/time_per_step_avg': '0.008839', 'train/epoch_time_elapsed': '169', 'train/estimated_remaining_minutes': '0.3235'}
|
| 302 |
+
{'loss': '3.643', 'grad_norm': '1.375', 'learning_rate': '0.0003', 'epoch': '0.04887', 'train/total_time_seconds': '26.55', 'train/time_per_step_avg': '0.008835', 'train/epoch_time_elapsed': '169.5', 'train/estimated_remaining_minutes': '0.3204'}
|
| 303 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 304 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 305 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 306 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 364.25it/s]
|
| 307 |
+
60%|██████ | 3000/5000 [02:59<00:43, 46.4[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 308 |
+
{'loss': '3.635', 'grad_norm': '1.812', 'learning_rate': '0.0003', 'epoch': '0.04921', 'train/total_time_seconds': '26.72', 'train/time_per_step_avg': '0.008811', 'train/epoch_time_elapsed': '169.9', 'train/estimated_remaining_minutes': '0.3173'}
|
| 309 |
+
{'loss': '3.661', 'grad_norm': '1.492', 'learning_rate': '0.0003', 'epoch': '0.04954', 'train/total_time_seconds': '26.9', 'train/time_per_step_avg': '0.008801', 'train/epoch_time_elapsed': '170.3', 'train/estimated_remaining_minutes': '0.3141'}
|
| 310 |
+
{'loss': '3.663', 'grad_norm': '1.445', 'learning_rate': '0.0003', 'epoch': '0.04988', 'train/total_time_seconds': '27.07', 'train/time_per_step_avg': '0.008791', 'train/epoch_time_elapsed': '170.8', 'train/estimated_remaining_minutes': '0.311'}
|
| 311 |
+
{'loss': '3.671', 'grad_norm': '1.547', 'learning_rate': '0.0003', 'epoch': '0.05022', 'train/total_time_seconds': '27.25', 'train/time_per_step_avg': '0.008756', 'train/epoch_time_elapsed': '171.2', 'train/estimated_remaining_minutes': '0.3078'}
|
| 312 |
+
{'loss': '3.639', 'grad_norm': '1.68', 'learning_rate': '0.0003', 'epoch': '0.05056', 'train/total_time_seconds': '27.42', 'train/time_per_step_avg': '0.008761', 'train/epoch_time_elapsed': '171.6', 'train/estimated_remaining_minutes': '0.3047'}
|
| 313 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 314 |
+
{'eval_loss': '3.654', 'eval_runtime': '7.423', 'eval_samples_per_second': '1283', 'eval_steps_per_second': '0.943', 'epoch': '0.05056', 'train/total_time_seconds': '27.42', 'train/time_per_step_avg': '0.008761', 'train/epoch_time_elapsed': '179', 'train/estimated_remaining_minutes': '0.3047'}
|
| 315 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 316 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 317 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 368.02it/s]
|
| 318 |
+
62%|██████▏ | 3100/5000 [03:01<00:40, 46.37it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 319 |
+
{'loss': '3.659', 'grad_norm': '1.453', 'learning_rate': '0.0003', 'epoch': '0.05089', 'train/total_time_seconds': '27.64', 'train/time_per_step_avg': '0.009176', 'train/epoch_time_elapsed': '179.6', 'train/estimated_remaining_minutes': '0.302'}
|
| 320 |
+
{'loss': '3.64', 'grad_norm': '1.648', 'learning_rate': '0.0003', 'epoch': '0.05123', 'train/total_time_seconds': '27.81', 'train/time_per_step_avg': '0.00917', 'train/epoch_time_elapsed': '180', 'train/estimated_remaining_minutes': '0.2989'}
|
| 321 |
+
{'loss': '3.672', 'grad_norm': '1.477', 'learning_rate': '0.0003', 'epoch': '0.05157', 'train/total_time_seconds': '27.99', 'train/time_per_step_avg': '0.009162', 'train/epoch_time_elapsed': '180.4', 'train/estimated_remaining_minutes': '0.2957'}
|
| 322 |
+
{'loss': '3.65', 'grad_norm': '1.43', 'learning_rate': '0.0003', 'epoch': '0.0519', 'train/total_time_seconds': '28.16', 'train/time_per_step_avg': '0.009174', 'train/epoch_time_elapsed': '180.9', 'train/estimated_remaining_minutes': '0.2926'}
|
| 323 |
+
{'loss': '3.689', 'grad_norm': '1.797', 'learning_rate': '0.0003', 'epoch': '0.05224', 'train/total_time_seconds': '28.34', 'train/time_per_step_avg': '0.009164', 'train/epoch_time_elapsed': '181.3', 'train/estimated_remaining_minutes': '0.2895'}
|
| 324 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 325 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 326 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 327 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 367.66it/s]
|
| 328 |
+
64%|██████▍ | 3200/5000 [03:11<00:38, 46.8[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 329 |
+
{'loss': '3.639', 'grad_norm': '1.711', 'learning_rate': '0.0003', 'epoch': '0.05258', 'train/total_time_seconds': '28.53', 'train/time_per_step_avg': '0.008881', 'train/epoch_time_elapsed': '181.8', 'train/estimated_remaining_minutes': '0.2865'}
|
| 330 |
+
{'loss': '3.633', 'grad_norm': '1.828', 'learning_rate': '0.0003', 'epoch': '0.05292', 'train/total_time_seconds': '28.7', 'train/time_per_step_avg': '0.008877', 'train/epoch_time_elapsed': '182.2', 'train/estimated_remaining_minutes': '0.2834'}
|
| 331 |
+
{'loss': '3.645', 'grad_norm': '1.461', 'learning_rate': '0.0003', 'epoch': '0.05325', 'train/total_time_seconds': '28.88', 'train/time_per_step_avg': '0.008892', 'train/epoch_time_elapsed': '182.6', 'train/estimated_remaining_minutes': '0.2803'}
|
| 332 |
+
{'loss': '3.639', 'grad_norm': '1.75', 'learning_rate': '0.0003', 'epoch': '0.05359', 'train/total_time_seconds': '29.06', 'train/time_per_step_avg': '0.008942', 'train/epoch_time_elapsed': '183.1', 'train/estimated_remaining_minutes': '0.2772'}
|
| 333 |
+
{'loss': '3.639', 'grad_norm': '1.398', 'learning_rate': '0.0003', 'epoch': '0.05393', 'train/total_time_seconds': '29.23', 'train/time_per_step_avg': '0.008957', 'train/epoch_time_elapsed': '183.5', 'train/estimated_remaining_minutes': '0.2741'}
|
| 334 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 335 |
+
{'eval_loss': '3.632', 'eval_runtime': '7.634', 'eval_samples_per_second': '1248', 'eval_steps_per_second': '0.917', 'epoch': '0.05393', 'train/total_time_seconds': '29.23', 'train/time_per_step_avg': '0.008957', 'train/epoch_time_elapsed': '191.1', 'train/estimated_remaining_minutes': '0.2741'}
|
| 336 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 337 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 338 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 231.50it/s]
|
| 339 |
+
66%|██████▌ | 3300/5000 [03:13<00:37, 45.49it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 340 |
+
{'loss': '3.656', 'grad_norm': '1.43', 'learning_rate': '0.0003', 'epoch': '0.05426', 'train/total_time_seconds': '29.41', 'train/time_per_step_avg': '0.00882', 'train/epoch_time_elapsed': '191.6', 'train/estimated_remaining_minutes': '0.271'}
|
| 341 |
+
{'loss': '3.648', 'grad_norm': '1.664', 'learning_rate': '0.0003', 'epoch': '0.0546', 'train/total_time_seconds': '29.59', 'train/time_per_step_avg': '0.008838', 'train/epoch_time_elapsed': '192', 'train/estimated_remaining_minutes': '0.2679'}
|
| 342 |
+
{'loss': '3.649', 'grad_norm': '1.891', 'learning_rate': '0.0003', 'epoch': '0.05494', 'train/total_time_seconds': '29.76', 'train/time_per_step_avg': '0.008847', 'train/epoch_time_elapsed': '192.5', 'train/estimated_remaining_minutes': '0.2648'}
|
| 343 |
+
{'loss': '3.615', 'grad_norm': '1.5', 'learning_rate': '0.0003', 'epoch': '0.05527', 'train/total_time_seconds': '29.94', 'train/time_per_step_avg': '0.008814', 'train/epoch_time_elapsed': '192.9', 'train/estimated_remaining_minutes': '0.2617'}
|
| 344 |
+
{'loss': '3.658', 'grad_norm': '1.453', 'learning_rate': '0.0003', 'epoch': '0.05561', 'train/total_time_seconds': '30.12', 'train/time_per_step_avg': '0.008825', 'train/epoch_time_elapsed': '193.3', 'train/estimated_remaining_minutes': '0.2586'}
|
| 345 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 346 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 347 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 348 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 265.11it/s]
|
| 349 |
+
68%|██████▊ | 3400/5000 [03:23<00:34, 46.3[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 350 |
+
{'loss': '3.633', 'grad_norm': '1.5', 'learning_rate': '0.0003', 'epoch': '0.05595', 'train/total_time_seconds': '30.29', 'train/time_per_step_avg': '0.00883', 'train/epoch_time_elapsed': '193.8', 'train/estimated_remaining_minutes': '0.2555'}
|
| 351 |
+
{'loss': '3.608', 'grad_norm': '1.898', 'learning_rate': '0.0003', 'epoch': '0.05629', 'train/total_time_seconds': '30.47', 'train/time_per_step_avg': '0.008815', 'train/epoch_time_elapsed': '194.2', 'train/estimated_remaining_minutes': '0.2524'}
|
| 352 |
+
{'loss': '3.628', 'grad_norm': '1.547', 'learning_rate': '0.0003', 'epoch': '0.05662', 'train/total_time_seconds': '30.64', 'train/time_per_step_avg': '0.008813', 'train/epoch_time_elapsed': '194.6', 'train/estimated_remaining_minutes': '0.2493'}
|
| 353 |
+
{'loss': '3.599', 'grad_norm': '1.43', 'learning_rate': '0.0003', 'epoch': '0.05696', 'train/total_time_seconds': '30.82', 'train/time_per_step_avg': '0.00884', 'train/epoch_time_elapsed': '195.1', 'train/estimated_remaining_minutes': '0.2462'}
|
| 354 |
+
{'loss': '3.626', 'grad_norm': '1.273', 'learning_rate': '0.0003', 'epoch': '0.0573', 'train/total_time_seconds': '31', 'train/time_per_step_avg': '0.008868', 'train/epoch_time_elapsed': '195.5', 'train/estimated_remaining_minutes': '0.2432'}
|
| 355 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 356 |
+
{'eval_loss': '3.615', 'eval_runtime': '7.519', 'eval_samples_per_second': '1267', 'eval_steps_per_second': '0.931', 'epoch': '0.0573', 'train/total_time_seconds': '31', 'train/time_per_step_avg': '0.008868', 'train/epoch_time_elapsed': '203', 'train/estimated_remaining_minutes': '0.2432'}
|
| 357 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 358 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 359 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 356.45it/s]
|
| 360 |
+
70%|██████▉ | 3487/5000 [03:24<00:34, 44.45it/s], ?it/s]
|
| 361 |
+
{'loss': '3.591', 'grad_norm': '1.336', 'learning_rate': '0.0003', 'epoch': '0.05763', 'train/total_time_seconds': '31.19', 'train/time_per_step_avg': '0.008951', 'train/epoch_time_elapsed': '203.5', 'train/estimated_remaining_minutes': '0.2401'}
|
| 362 |
+
{'loss': '3.62', 'grad_norm': '1.562', 'learning_rate': '0.0003', 'epoch': '0.05797', 'train/total_time_seconds': '31.37', 'train/time_per_step_avg': '0.008993', 'train/epoch_time_elapsed': '203.9', 'train/estimated_remaining_minutes': '0.2371'}
|
| 363 |
+
{'loss': '3.599', 'grad_norm': '1.773', 'learning_rate': '0.0003', 'epoch': '0.05831', 'train/total_time_seconds': '31.55', 'train/time_per_step_avg': '0.009062', 'train/epoch_time_elapsed': '204.3', 'train/estimated_remaining_minutes': '0.234'}
|
| 364 |
+
{'loss': '3.616', 'grad_norm': '1.57', 'learning_rate': '0.0003', 'epoch': '0.05865', 'train/total_time_seconds': '31.73', 'train/time_per_step_avg': '0.00904', 'train/epoch_time_elapsed': '204.8', 'train/estimated_remaining_minutes': '0.231'}
|
zain/Activation/wandb/run-20260819_163014-smtfejrp/files/requirements.txt
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
asttokens==3.0.1
|
| 2 |
+
comm==0.2.3
|
| 3 |
+
debugpy==1.8.21
|
| 4 |
+
decorator==5.3.1
|
| 5 |
+
executing==2.2.1
|
| 6 |
+
nest-asyncio==1.6.0
|
| 7 |
+
parso==0.8.7
|
| 8 |
+
platformdirs==4.11.0
|
| 9 |
+
psutil==7.2.2
|
| 10 |
+
ptyprocess==0.7.0
|
| 11 |
+
pure_eval==0.2.3
|
| 12 |
+
Pygments==2.20.0
|
| 13 |
+
pyzmq==27.1.0
|
| 14 |
+
setuptools==83.0.0
|
| 15 |
+
six==1.17.0
|
| 16 |
+
tornado==6.5.7
|
| 17 |
+
traitlets==5.15.0
|
| 18 |
+
fsspec==2026.4.0
|
| 19 |
+
wcwidth==0.8.2
|
| 20 |
+
ipython_pygments_lexers==1.1.1
|
| 21 |
+
jedi==0.20.0
|
| 22 |
+
jupyter_core==5.9.1
|
| 23 |
+
matplotlib-inline==0.2.2
|
| 24 |
+
pexpect==4.9.0
|
| 25 |
+
prompt_toolkit==3.0.53
|
| 26 |
+
python-dateutil==2.9.0.post0
|
| 27 |
+
stack_data==0.6.3
|
| 28 |
+
wheel==0.47.0
|
| 29 |
+
jupyter_client==8.9.1
|
| 30 |
+
pip==26.1.2
|
| 31 |
+
ipython==9.15.0
|
| 32 |
+
ipykernel==7.2.0
|
| 33 |
+
threadpoolctl==3.6.0
|
| 34 |
+
pyparsing==3.3.2
|
| 35 |
+
typing_extensions==4.15.0
|
| 36 |
+
Jinja2==3.1.6
|
| 37 |
+
narwhals==2.24.0
|
| 38 |
+
kiwisolver==1.5.0
|
| 39 |
+
joblib==1.5.3
|
| 40 |
+
fonttools==4.63.0
|
| 41 |
+
cycler==0.12.1
|
| 42 |
+
scipy==1.17.1
|
| 43 |
+
pandas==3.0.5
|
| 44 |
+
contourpy==1.3.3
|
| 45 |
+
scikit-learn==1.9.0
|
| 46 |
+
matplotlib==3.11.1
|
| 47 |
+
urllib3==2.7.0
|
| 48 |
+
tqdm==4.70.0
|
| 49 |
+
idna==3.18
|
| 50 |
+
charset-normalizer==3.4.9
|
| 51 |
+
certifi==2026.7.22
|
| 52 |
+
requests==2.34.2
|
| 53 |
+
seaborn==0.13.2
|
| 54 |
+
uv==0.12.0
|
| 55 |
+
shellingham==1.5.4
|
| 56 |
+
mpmath==1.3.0
|
| 57 |
+
attrs==26.1.0
|
| 58 |
+
hf-xet==1.5.2
|
| 59 |
+
nvidia-nccl-cu12==2.21.5
|
| 60 |
+
MarkupSafe==3.0.3
|
| 61 |
+
regex==2026.7.19
|
| 62 |
+
importlib_metadata==9.0.0
|
| 63 |
+
httpcore==1.0.9
|
| 64 |
+
annotated-doc==0.0.5
|
| 65 |
+
multidict==6.7.1
|
| 66 |
+
aiohttp==3.14.3
|
| 67 |
+
aiosignal==1.4.0
|
| 68 |
+
xxhash==3.8.1
|
| 69 |
+
aiohappyeyeballs==2.7.1
|
| 70 |
+
mdurl==0.1.2
|
| 71 |
+
cuda-toolkit==13.0.3.0
|
| 72 |
+
networkx==3.6.1
|
| 73 |
+
PyYAML==6.0.3
|
| 74 |
+
nvidia-cufile==1.15.1.6
|
| 75 |
+
typer==0.27.0
|
| 76 |
+
torchaudio==2.6.0+cu124
|
| 77 |
+
rich==15.0.0
|
| 78 |
+
nvidia-cufft-cu12==11.2.1.3
|
| 79 |
+
h11==0.16.0
|
| 80 |
+
dill==0.4.1
|
| 81 |
+
cuda-pathfinder==1.6.0
|
| 82 |
+
filelock==3.29.0
|
| 83 |
+
nvidia-nvtx-cu12==12.4.127
|
| 84 |
+
httpx==0.28.1
|
| 85 |
+
anyio==4.14.2
|
| 86 |
+
numpy==2.4.4
|
| 87 |
+
yarl==1.24.5
|
| 88 |
+
click==8.4.2
|
| 89 |
+
triton==3.2.0
|
| 90 |
+
frozenlist==1.8.0
|
| 91 |
+
zipp==4.1.0
|
| 92 |
+
propcache==0.5.2
|
| 93 |
+
markdown-it-py==4.2.0
|
| 94 |
+
nvidia-cuda-runtime==13.0.96
|
| 95 |
+
cuda-bindings==13.3.1
|
| 96 |
+
nvidia-cuda-cupti==13.0.85
|
| 97 |
+
torch==2.6.0+cu124
|
| 98 |
+
multiprocess==0.70.19
|
| 99 |
+
pillow==12.2.0
|
| 100 |
+
transformers==5.16.0.dev0
|
| 101 |
+
wandb==0.28.1
|
| 102 |
+
nvidia-curand==10.4.0.35
|
| 103 |
+
sympy==1.13.1
|
| 104 |
+
nvidia-cusparse==12.6.3.3
|
| 105 |
+
nvidia-cuda-nvrtc==13.0.88
|
| 106 |
+
typing-inspection==0.4.2
|
| 107 |
+
nvidia-cusolver==12.0.4.66
|
| 108 |
+
nvidia-cufft==12.0.0.61
|
| 109 |
+
nvidia-cudnn-cu13==9.20.0.48
|
| 110 |
+
nvidia-cublas==13.1.1.3
|
| 111 |
+
pyarrow==25.0.0
|
| 112 |
+
evaluate==0.4.6
|
| 113 |
+
diffusers==0.39.0
|
| 114 |
+
pydantic==2.13.4
|
| 115 |
+
annotated-types==0.8.0
|
| 116 |
+
protobuf==7.35.1
|
| 117 |
+
sentry-sdk==2.66.1
|
| 118 |
+
einops==0.8.2
|
| 119 |
+
packaging==26.2
|
| 120 |
+
nvidia-nvjitlink-cu12==12.4.127
|
| 121 |
+
nvidia-curand-cu12==10.3.5.147
|
| 122 |
+
nvidia-cusparselt-cu12==0.6.2
|
| 123 |
+
nvidia-cusparse-cu12==12.3.1.170
|
| 124 |
+
nvidia-cuda-runtime-cu12==12.4.127
|
| 125 |
+
torchvision==0.21.0+cu124
|
| 126 |
+
nvidia-cuda-nvrtc-cu12==12.4.127
|
| 127 |
+
nvidia-cuda-cupti-cu12==12.4.127
|
| 128 |
+
nvidia-cusolver-cu12==11.6.1.9
|
| 129 |
+
nvidia-cublas-cu12==12.4.5.8
|
| 130 |
+
nvidia-cudnn-cu12==9.1.0.70
|
| 131 |
+
huggingface_hub==1.26.0
|
| 132 |
+
datasets==5.0.1
|
| 133 |
+
safetensors==0.8.0
|
| 134 |
+
accelerate==1.14.0
|
| 135 |
+
pydantic_core==2.46.4
|
| 136 |
+
ninja==1.13.0
|
| 137 |
+
tokenizers==0.23.1
|
| 138 |
+
autocommand==2.2.2
|
| 139 |
+
backports.tarfile==1.2.0
|
| 140 |
+
importlib_metadata==8.7.1
|
| 141 |
+
jaraco.text==4.0.0
|
| 142 |
+
jaraco.context==6.1.0
|
| 143 |
+
jaraco.functools==4.4.0
|
| 144 |
+
more-itertools==10.8.0
|
| 145 |
+
packaging==26.0
|
| 146 |
+
platformdirs==4.4.0
|
| 147 |
+
tomli==2.4.0
|
| 148 |
+
wheel==0.46.3
|
| 149 |
+
zipp==3.23.0
|
zain/Activation/wandb/run-20260819_163014-smtfejrp/files/wandb-metadata.json
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"os": "Linux-5.15.0-126-generic-x86_64-with-glibc2.35",
|
| 3 |
+
"python": "CPython 3.11.15",
|
| 4 |
+
"startedAt": "2026-08-19T16:30:14.405953Z",
|
| 5 |
+
"args": [
|
| 6 |
+
"--config",
|
| 7 |
+
"/mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/configs/baseline100L.yaml",
|
| 8 |
+
"--variants",
|
| 9 |
+
"mlp-linear-3L",
|
| 10 |
+
"mlp-gelu-3L",
|
| 11 |
+
"mlp-relu-3L",
|
| 12 |
+
"mlp-silu-waleed10-3L",
|
| 13 |
+
"mlp-waleed10-3L",
|
| 14 |
+
"mlp-s10-3L",
|
| 15 |
+
"mlp-w1a-3L",
|
| 16 |
+
"mlp-silu-3L",
|
| 17 |
+
"mlp-sigmoid-3L",
|
| 18 |
+
"mlp-tanh-3L",
|
| 19 |
+
"glu-linear-3L",
|
| 20 |
+
"glu-gelu-3L",
|
| 21 |
+
"glu-relu-3L",
|
| 22 |
+
"glu-powlu-3L",
|
| 23 |
+
"glu-situglu-3L",
|
| 24 |
+
"glu-waleed-3L",
|
| 25 |
+
"glu-waleedglu_low-3L",
|
| 26 |
+
"glu-situglu_low-3L",
|
| 27 |
+
"glu-silu-waleed10-3L",
|
| 28 |
+
"glu-waleed10-3L",
|
| 29 |
+
"glu-w1a-3L",
|
| 30 |
+
"glu-silu-3L",
|
| 31 |
+
"glu-sigmoid-3L",
|
| 32 |
+
"glu-tanh-3L"
|
| 33 |
+
],
|
| 34 |
+
"program": "/mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/sweep.py",
|
| 35 |
+
"codePath": "sweep.py",
|
| 36 |
+
"codePathLocal": "sweep.py",
|
| 37 |
+
"git": {
|
| 38 |
+
"remote": "https://github.com/w-ahmad1a10/Activation.git",
|
| 39 |
+
"commit": "463c9961366755fc55f02df9a0d471b3cbcb025e"
|
| 40 |
+
},
|
| 41 |
+
"email": "deepnevro@gmail.com",
|
| 42 |
+
"root": "/mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation",
|
| 43 |
+
"host": "deeplens-k3s-node1",
|
| 44 |
+
"executable": "/mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python",
|
| 45 |
+
"cpu_count": 112,
|
| 46 |
+
"cpu_count_logical": 224,
|
| 47 |
+
"gpu": "NVIDIA H100 80GB HBM3",
|
| 48 |
+
"gpu_count": 8,
|
| 49 |
+
"disk": {
|
| 50 |
+
"/": {
|
| 51 |
+
"total": "1560765693952",
|
| 52 |
+
"used": "708391636992"
|
| 53 |
+
}
|
| 54 |
+
},
|
| 55 |
+
"memory": {
|
| 56 |
+
"total": "2164089937920"
|
| 57 |
+
},
|
| 58 |
+
"gpu_nvidia": [
|
| 59 |
+
{
|
| 60 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 61 |
+
"memoryTotal": "85520809984",
|
| 62 |
+
"cudaCores": 16896,
|
| 63 |
+
"architecture": "Hopper",
|
| 64 |
+
"uuid": "GPU-39c684a5-fde6-83d7-1663-0859795881ae"
|
| 65 |
+
},
|
| 66 |
+
{
|
| 67 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 68 |
+
"memoryTotal": "85520809984",
|
| 69 |
+
"cudaCores": 16896,
|
| 70 |
+
"architecture": "Hopper",
|
| 71 |
+
"uuid": "GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3"
|
| 72 |
+
},
|
| 73 |
+
{
|
| 74 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 75 |
+
"memoryTotal": "85520809984",
|
| 76 |
+
"cudaCores": 16896,
|
| 77 |
+
"architecture": "Hopper",
|
| 78 |
+
"uuid": "GPU-132944c4-b689-2b5f-89a4-d730401677ab"
|
| 79 |
+
},
|
| 80 |
+
{
|
| 81 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 82 |
+
"memoryTotal": "85520809984",
|
| 83 |
+
"cudaCores": 16896,
|
| 84 |
+
"architecture": "Hopper",
|
| 85 |
+
"uuid": "GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864"
|
| 86 |
+
},
|
| 87 |
+
{
|
| 88 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 89 |
+
"memoryTotal": "85520809984",
|
| 90 |
+
"cudaCores": 16896,
|
| 91 |
+
"architecture": "Hopper",
|
| 92 |
+
"uuid": "GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef"
|
| 93 |
+
},
|
| 94 |
+
{
|
| 95 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 96 |
+
"memoryTotal": "85520809984",
|
| 97 |
+
"cudaCores": 16896,
|
| 98 |
+
"architecture": "Hopper",
|
| 99 |
+
"uuid": "GPU-bc6c3e3c-9b90-09ca-c034-774961847c54"
|
| 100 |
+
},
|
| 101 |
+
{
|
| 102 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 103 |
+
"memoryTotal": "85520809984",
|
| 104 |
+
"cudaCores": 16896,
|
| 105 |
+
"architecture": "Hopper",
|
| 106 |
+
"uuid": "GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9"
|
| 107 |
+
},
|
| 108 |
+
{
|
| 109 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 110 |
+
"memoryTotal": "85520809984",
|
| 111 |
+
"cudaCores": 16896,
|
| 112 |
+
"architecture": "Hopper",
|
| 113 |
+
"uuid": "GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea"
|
| 114 |
+
}
|
| 115 |
+
],
|
| 116 |
+
"cudaVersion": "12.4",
|
| 117 |
+
"writerId": "llfkq1g0g72x310twkfxxx7wkoxj7gtx"
|
| 118 |
+
}
|