Auto upload zain 2026-08-14T23:07:46.851605 (part 6)
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- zain/Activation/out/glu-waleedglu_low-21L_run/tokenizer_config.json +13 -0
- zain/Activation/out/glu-waleedglu_low-21L_run/training_args.bin +3 -0
- zain/Activation/out/glu-waleedglu_low-21L_run/training_log.jsonl +53 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/config.json +36 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/model.safetensors +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/optimizer.pt +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/rng_state.pth +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/scheduler.pt +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/tokenizer.json +0 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/tokenizer_config.json +13 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/trainer_state.json +69 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/training_args.bin +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/config.json +36 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/model.safetensors +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/optimizer.pt +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/rng_state.pth +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/scheduler.pt +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/tokenizer.json +0 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/tokenizer_config.json +13 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/trainer_state.json +392 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/training_args.bin +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/config.json +36 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/model.safetensors +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/optimizer.pt +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/rng_state.pth +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/scheduler.pt +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/tokenizer.json +0 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/tokenizer_config.json +13 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/trainer_state.json +104 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/training_args.bin +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/config.json +36 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/model.safetensors +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/optimizer.pt +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/rng_state.pth +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/scheduler.pt +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/tokenizer.json +0 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/tokenizer_config.json +13 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/trainer_state.json +139 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/training_args.bin +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/config.json +36 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/model.safetensors +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/optimizer.pt +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/rng_state.pth +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/scheduler.pt +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/tokenizer.json +0 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/tokenizer_config.json +13 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/trainer_state.json +174 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/training_args.bin +3 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-500/config.json +36 -0
- zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-500/model.safetensors +3 -0
zain/Activation/out/glu-waleedglu_low-21L_run/tokenizer_config.json
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"backend": "tokenizers",
|
| 4 |
+
"bos_token": "<|endoftext|>",
|
| 5 |
+
"eos_token": "<|endoftext|>",
|
| 6 |
+
"errors": "replace",
|
| 7 |
+
"is_local": false,
|
| 8 |
+
"local_files_only": false,
|
| 9 |
+
"model_max_length": 1024,
|
| 10 |
+
"pad_token": "<|endoftext|>",
|
| 11 |
+
"tokenizer_class": "GPT2Tokenizer",
|
| 12 |
+
"unk_token": "<|endoftext|>"
|
| 13 |
+
}
|
zain/Activation/out/glu-waleedglu_low-21L_run/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b710bd96acfb6390e443452d6d294cbaf327cdd790a050acb04137ff8f2a9d50
|
| 3 |
+
size 4920
|
zain/Activation/out/glu-waleedglu_low-21L_run/training_log.jsonl
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"step": 20, "epoch": 0.010784578053383662, "timestamp": 1786747955.5507054, "loss": 28.87822570800781, "grad_norm": 4.0, "learning_rate": 0.001, "train/total_time_seconds": 8.125255100429058, "train/time_per_step_avg": 0.4062627550214529, "train/epoch_time_elapsed": 15.911321021616459, "train/estimated_remaining_minutes": 6.63562499868373}
|
| 2 |
+
{"step": 40, "epoch": 0.021569156106767323, "timestamp": 1786747971.392193, "loss": 24.06096496582031, "grad_norm": 0.921875, "learning_rate": 0.001, "train/total_time_seconds": 16.25177477672696, "train/time_per_step_avg": 0.40629436941817404, "train/epoch_time_elapsed": 31.752809405326843, "train/estimated_remaining_minutes": 6.500709910690785}
|
| 3 |
+
{"step": 60, "epoch": 0.032353734160150985, "timestamp": 1786747987.2228582, "loss": 22.317706298828124, "grad_norm": 1.4140625, "learning_rate": 0.001, "train/total_time_seconds": 24.36207727342844, "train/time_per_step_avg": 0.40603462122380735, "train/epoch_time_elapsed": 47.58347408473492, "train/estimated_remaining_minutes": 6.361209065839648}
|
| 4 |
+
{"step": 80, "epoch": 0.043138312213534646, "timestamp": 1786748002.850897, "loss": 20.63355407714844, "grad_norm": 2.484375, "learning_rate": 0.001, "train/total_time_seconds": 32.44940710440278, "train/time_per_step_avg": 0.40561758880503473, "train/epoch_time_elapsed": 63.2115133702755, "train/estimated_remaining_minutes": 6.219469695010533}
|
| 5 |
+
{"step": 100, "epoch": 0.05392289026691831, "timestamp": 1786748018.7311933, "loss": 18.89258575439453, "grad_norm": 1.8515625, "learning_rate": 0.001, "train/total_time_seconds": 40.5365272462368, "train/time_per_step_avg": 0.40536527246236803, "train/epoch_time_elapsed": 79.09180996194482, "train/estimated_remaining_minutes": 6.0804790869355205}
|
| 6 |
+
{"step": 120, "epoch": 0.06470746832030197, "timestamp": 1786748034.5811245, "loss": 17.514393615722657, "grad_norm": 2.140625, "learning_rate": 0.001, "train/total_time_seconds": 48.669238332659006, "train/time_per_step_avg": 0.40543983232229946, "train/epoch_time_elapsed": 94.94174122810364, "train/estimated_remaining_minutes": 5.948462462880545}
|
| 7 |
+
{"step": 140, "epoch": 0.07549204637368563, "timestamp": 1786748050.41628, "loss": 16.459072875976563, "grad_norm": 2.671875, "learning_rate": 0.001, "train/total_time_seconds": 56.77880089357495, "train/time_per_step_avg": 0.4052702611684799, "train/epoch_time_elapsed": 110.77689604461193, "train/estimated_remaining_minutes": 5.813067710532674}
|
| 8 |
+
{"step": 160, "epoch": 0.08627662442706929, "timestamp": 1786748066.2085338, "loss": 15.619828796386718, "grad_norm": 2.34375, "learning_rate": 0.001, "train/total_time_seconds": 64.8743740580976, "train/time_per_step_avg": 0.4051229678466916, "train/epoch_time_elapsed": 126.56915009394288, "train/estimated_remaining_minutes": 5.67650773008354}
|
| 9 |
+
{"step": 180, "epoch": 0.09706120248045295, "timestamp": 1786748082.0450277, "loss": 14.93470458984375, "grad_norm": 2.125, "learning_rate": 0.001, "train/total_time_seconds": 73.00178475677967, "train/time_per_step_avg": 0.4055237765237689, "train/epoch_time_elapsed": 142.4056449122727, "train/estimated_remaining_minutes": 5.5427281019036405}
|
| 10 |
+
{"step": 200, "epoch": 0.10784578053383662, "timestamp": 1786748097.7236798, "loss": 14.343766784667968, "grad_norm": 3.703125, "learning_rate": 0.001, "train/total_time_seconds": 81.08889551833272, "train/time_per_step_avg": 0.4055236827209592, "train/epoch_time_elapsed": 158.0842955261469, "train/estimated_remaining_minutes": 5.405926367888848}
|
| 11 |
+
{"step": 220, "epoch": 0.11863035858722028, "timestamp": 1786748113.5104978, "loss": 13.889820861816407, "grad_norm": 2.1875, "learning_rate": 0.001, "train/total_time_seconds": 89.17833111807704, "train/time_per_step_avg": 0.4050909278541803, "train/epoch_time_elapsed": 173.8711139522493, "train/estimated_remaining_minutes": 5.26962865697728}
|
| 12 |
+
{"step": 240, "epoch": 0.12941493664060394, "timestamp": 1786748129.3850553, "loss": 13.443467712402343, "grad_norm": 2.9375, "learning_rate": 0.001, "train/total_time_seconds": 97.44434486329556, "train/time_per_step_avg": 0.406655439697206, "train/epoch_time_elapsed": 189.74567170068622, "train/estimated_remaining_minutes": 5.142895978896154}
|
| 13 |
+
{"step": 260, "epoch": 0.1401995146939876, "timestamp": 1786748145.3215604, "loss": 12.998054504394531, "grad_norm": 1.6484375, "learning_rate": 0.001, "train/total_time_seconds": 105.56989968568087, "train/time_per_step_avg": 0.4069552562758327, "train/epoch_time_elapsed": 205.68217654898763, "train/estimated_remaining_minutes": 5.007802933807938}
|
| 14 |
+
{"step": 280, "epoch": 0.15098409274737126, "timestamp": 1786748161.172112, "loss": 12.626632690429688, "grad_norm": 2.453125, "learning_rate": 0.001, "train/total_time_seconds": 113.67667726427317, "train/time_per_step_avg": 0.406748925074935, "train/epoch_time_elapsed": 221.53272734954953, "train/estimated_remaining_minutes": 4.871857597040279}
|
| 15 |
+
{"step": 300, "epoch": 0.1617686708007549, "timestamp": 1786748176.9442291, "loss": 12.287302398681641, "grad_norm": 2.46875, "learning_rate": 0.001, "train/total_time_seconds": 121.78035191074014, "train/time_per_step_avg": 0.40691456392407416, "train/epoch_time_elapsed": 237.30484534427524, "train/estimated_remaining_minutes": 4.735902574306562}
|
| 16 |
+
{"step": 320, "epoch": 0.17255324885413859, "timestamp": 1786748192.868352, "loss": 11.972341156005859, "grad_norm": 1.765625, "learning_rate": 0.001, "train/total_time_seconds": 129.87322784215212, "train/time_per_step_avg": 0.4069489672407508, "train/epoch_time_elapsed": 253.228967346251, "train/estimated_remaining_minutes": 4.599676819409554}
|
| 17 |
+
{"step": 340, "epoch": 0.18333782690752223, "timestamp": 1786748208.6168966, "loss": 11.687677764892578, "grad_norm": 2.359375, "learning_rate": 0.001, "train/total_time_seconds": 137.99583918601274, "train/time_per_step_avg": 0.4055149432271719, "train/epoch_time_elapsed": 268.97751319408417, "train/estimated_remaining_minutes": 4.464571267782765}
|
| 18 |
+
{"step": 360, "epoch": 0.1941224049609059, "timestamp": 1786748224.4517117, "loss": 11.4271728515625, "grad_norm": 2.078125, "learning_rate": 0.001, "train/total_time_seconds": 146.12659545615315, "train/time_per_step_avg": 0.4055669577047229, "train/epoch_time_elapsed": 284.812327362597, "train/estimated_remaining_minutes": 4.329676902404538}
|
| 19 |
+
{"step": 380, "epoch": 0.20490698301428956, "timestamp": 1786748240.3111615, "loss": 11.163407135009766, "grad_norm": 2.0, "learning_rate": 0.001, "train/total_time_seconds": 154.2320696786046, "train/time_per_step_avg": 0.40555392414331437, "train/epoch_time_elapsed": 300.67177828401327, "train/estimated_remaining_minutes": 4.194029964944511}
|
| 20 |
+
{"step": 400, "epoch": 0.21569156106767323, "timestamp": 1786748256.2138653, "loss": 10.960685729980469, "grad_norm": 1.9453125, "learning_rate": 0.001, "train/total_time_seconds": 162.36926352605224, "train/time_per_step_avg": 0.405889116153121, "train/epoch_time_elapsed": 316.57448191568255, "train/estimated_remaining_minutes": 4.059231588151306}
|
| 21 |
+
{"step": 420, "epoch": 0.22647613912105688, "timestamp": 1786748272.037616, "loss": 10.743543243408203, "grad_norm": 1.9375, "learning_rate": 0.001, "train/total_time_seconds": 170.50094760209322, "train/time_per_step_avg": 0.406277197599411, "train/epoch_time_elapsed": 332.39823308214545, "train/estimated_remaining_minutes": 3.924228159095797}
|
| 22 |
+
{"step": 440, "epoch": 0.23726071717444056, "timestamp": 1786748287.8980882, "loss": 10.606502532958984, "grad_norm": 2.421875, "learning_rate": 0.001, "train/total_time_seconds": 178.61680870130658, "train/time_per_step_avg": 0.40620969515293837, "train/epoch_time_elapsed": 348.2587040960789, "train/estimated_remaining_minutes": 3.788841396694382}
|
| 23 |
+
{"step": 460, "epoch": 0.2480452952278242, "timestamp": 1786748303.573111, "loss": 10.407550048828124, "grad_norm": 2.375, "learning_rate": 0.001, "train/total_time_seconds": 186.72400549799204, "train/time_per_step_avg": 0.40597410041838883, "train/epoch_time_elapsed": 363.933727145195, "train/estimated_remaining_minutes": 3.6532957597433224}
|
| 24 |
+
{"step": 480, "epoch": 0.2588298732812079, "timestamp": 1786748319.429599, "loss": 10.243016052246094, "grad_norm": 2.109375, "learning_rate": 0.001, "train/total_time_seconds": 194.82527919858694, "train/time_per_step_avg": 0.4059320951998234, "train/epoch_time_elapsed": 379.79021544381976, "train/estimated_remaining_minutes": 3.5176786521967087}
|
| 25 |
+
{"step": 500, "epoch": 0.26961445133459155, "timestamp": 1786748335.175178, "loss": 10.094182586669922, "grad_norm": 1.953125, "learning_rate": 0.001, "train/total_time_seconds": 202.9594863653183, "train/time_per_step_avg": 0.4059022283926606, "train/epoch_time_elapsed": 395.5357943326235, "train/estimated_remaining_minutes": 3.3826581060886385}
|
| 26 |
+
{"step": 520, "epoch": 0.2803990293879752, "timestamp": 1786748351.1012967, "loss": 9.965821075439454, "grad_norm": 2.28125, "learning_rate": 0.001, "train/total_time_seconds": 211.05544440820813, "train/time_per_step_avg": 0.40554496806114915, "train/epoch_time_elapsed": 411.46191192790866, "train/estimated_remaining_minutes": 3.247006837049356}
|
| 27 |
+
{"step": 540, "epoch": 0.29118360744135885, "timestamp": 1786748366.8551178, "loss": 9.849958038330078, "grad_norm": 2.296875, "learning_rate": 0.001, "train/total_time_seconds": 219.1496572867036, "train/time_per_step_avg": 0.4053284858539701, "train/epoch_time_elapsed": 427.2157339602709, "train/estimated_remaining_minutes": 3.1113840232062855}
|
| 28 |
+
{"step": 560, "epoch": 0.3019681854947425, "timestamp": 1786748382.6123364, "loss": 9.730626678466797, "grad_norm": 2.140625, "learning_rate": 0.001, "train/total_time_seconds": 227.23986832424998, "train/time_per_step_avg": 0.4051586282625794, "train/epoch_time_elapsed": 442.9729521200061, "train/estimated_remaining_minutes": 2.9757601804366067}
|
| 29 |
+
{"step": 580, "epoch": 0.3127527635481262, "timestamp": 1786748398.3827922, "loss": 9.614877319335937, "grad_norm": 1.859375, "learning_rate": 0.001, "train/total_time_seconds": 235.33354591205716, "train/time_per_step_avg": 0.4050826671347022, "train/epoch_time_elapsed": 458.74340914189816, "train/estimated_remaining_minutes": 2.840232450662759}
|
| 30 |
+
{"step": 600, "epoch": 0.3235373416015098, "timestamp": 1786748414.3937943, "loss": 9.505727386474609, "grad_norm": 2.296875, "learning_rate": 0.001, "train/total_time_seconds": 243.46157986670732, "train/time_per_step_avg": 0.40502093501389025, "train/epoch_time_elapsed": 474.7544106952846, "train/estimated_remaining_minutes": 2.705128665185637}
|
| 31 |
+
{"step": 620, "epoch": 0.3343219196548935, "timestamp": 1786748430.427414, "loss": 9.414925384521485, "grad_norm": 2.046875, "learning_rate": 0.001, "train/total_time_seconds": 251.58637992665172, "train/time_per_step_avg": 0.40530935518443584, "train/epoch_time_elapsed": 490.7880291491747, "train/estimated_remaining_minutes": 2.569968397100206}
|
| 32 |
+
{"step": 640, "epoch": 0.34510649770827717, "timestamp": 1786748446.4542549, "loss": 9.328840637207032, "grad_norm": 1.9921875, "learning_rate": 0.001, "train/total_time_seconds": 259.7155615314841, "train/time_per_step_avg": 0.40565904244780543, "train/epoch_time_elapsed": 506.8148699365556, "train/estimated_remaining_minutes": 2.4348333893576637}
|
| 33 |
+
{"step": 660, "epoch": 0.35589107576166085, "timestamp": 1786748462.320152, "loss": 9.245575714111329, "grad_norm": 1.9609375, "learning_rate": 0.001, "train/total_time_seconds": 267.84740838781, "train/time_per_step_avg": 0.4060754006356001, "train/epoch_time_elapsed": 522.68076909706, "train/estimated_remaining_minutes": 2.2996999710064494}
|
| 34 |
+
{"step": 680, "epoch": 0.36667565381504447, "timestamp": 1786748478.2107358, "loss": 9.160669708251953, "grad_norm": 2.078125, "learning_rate": 0.001, "train/total_time_seconds": 275.9440933018923, "train/time_per_step_avg": 0.40610547389835117, "train/epoch_time_elapsed": 538.5713514313102, "train/estimated_remaining_minutes": 2.164267398446214}
|
| 35 |
+
{"step": 700, "epoch": 0.37746023186842814, "timestamp": 1786748493.9718537, "loss": 9.08907699584961, "grad_norm": 1.59375, "learning_rate": 0.001, "train/total_time_seconds": 284.0659934170544, "train/time_per_step_avg": 0.4060441355034709, "train/epoch_time_elapsed": 554.332470394671, "train/estimated_remaining_minutes": 2.029042810121817}
|
| 36 |
+
{"step": 720, "epoch": 0.3882448099218118, "timestamp": 1786748510.2042859, "loss": 8.998544311523437, "grad_norm": 1.921875, "learning_rate": 0.001, "train/total_time_seconds": 292.20687505602837, "train/time_per_step_avg": 0.4062049512937665, "train/epoch_time_elapsed": 570.5649027526379, "train/estimated_remaining_minutes": 1.8939334494372209}
|
| 37 |
+
{"step": 740, "epoch": 0.3990293879751955, "timestamp": 1786748526.0544066, "loss": 8.961595153808593, "grad_norm": 1.8515625, "learning_rate": 0.001, "train/total_time_seconds": 300.34223844110966, "train/time_per_step_avg": 0.4062667690962553, "train/epoch_time_elapsed": 586.4150231704116, "train/estimated_remaining_minutes": 1.7587608557362278}
|
| 38 |
+
{"step": 760, "epoch": 0.4098139660285791, "timestamp": 1786748541.9264393, "loss": 8.89373779296875, "grad_norm": 1.78125, "learning_rate": 0.001, "train/total_time_seconds": 308.4568819515407, "train/time_per_step_avg": 0.4060947356373072, "train/epoch_time_elapsed": 602.2870550639927, "train/estimated_remaining_minutes": 1.6234572734291617}
|
| 39 |
+
{"step": 780, "epoch": 0.4205985440819628, "timestamp": 1786748557.5973103, "loss": 8.858226776123047, "grad_norm": 1.90625, "learning_rate": 0.001, "train/total_time_seconds": 316.5688179396093, "train/time_per_step_avg": 0.4062472463771701, "train/epoch_time_elapsed": 617.9579257294536, "train/estimated_remaining_minutes": 1.4881440159554282}
|
| 40 |
+
{"step": 800, "epoch": 0.43138312213534646, "timestamp": 1786748573.3539927, "loss": 8.755474090576172, "grad_norm": 1.8359375, "learning_rate": 0.001, "train/total_time_seconds": 324.6625647544861, "train/time_per_step_avg": 0.4059657133743167, "train/epoch_time_elapsed": 633.7146084494889, "train/estimated_remaining_minutes": 1.3527606864770254}
|
| 41 |
+
{"step": 820, "epoch": 0.44216770018873014, "timestamp": 1786748589.0057478, "loss": 8.732923126220703, "grad_norm": 1.796875, "learning_rate": 0.001, "train/total_time_seconds": 332.75162014365196, "train/time_per_step_avg": 0.40544745087623596, "train/epoch_time_elapsed": 649.3663630895317, "train/estimated_remaining_minutes": 1.217383976135312}
|
| 42 |
+
{"step": 840, "epoch": 0.45295227824211376, "timestamp": 1786748604.9225793, "loss": 8.691609954833984, "grad_norm": 1.8203125, "learning_rate": 0.001, "train/total_time_seconds": 340.86094953864813, "train/time_per_step_avg": 0.40518711097538473, "train/epoch_time_elapsed": 665.2831956408918, "train/estimated_remaining_minutes": 1.0820982525036449}
|
| 43 |
+
{"step": 860, "epoch": 0.46373685629549743, "timestamp": 1786748620.8106012, "loss": 8.623181915283203, "grad_norm": 1.921875, "learning_rate": 0.001, "train/total_time_seconds": 348.9605983234942, "train/time_per_step_avg": 0.40503716371953485, "train/epoch_time_elapsed": 681.1712169833481, "train/estimated_remaining_minutes": 0.9467923210327363}
|
| 44 |
+
{"step": 880, "epoch": 0.4745214343488811, "timestamp": 1786748636.4933302, "loss": 8.585681915283203, "grad_norm": 1.84375, "learning_rate": 0.001, "train/total_time_seconds": 357.0373779311776, "train/time_per_step_avg": 0.40468559991568326, "train/epoch_time_elapsed": 696.8539464510977, "train/estimated_remaining_minutes": 0.8114485862072218}
|
| 45 |
+
{"step": 900, "epoch": 0.4853060124022648, "timestamp": 1786748652.084348, "loss": 8.534693145751953, "grad_norm": 1.9453125, "learning_rate": 0.001, "train/total_time_seconds": 365.11572790145874, "train/time_per_step_avg": 0.40453163146972654, "train/epoch_time_elapsed": 712.4449638240039, "train/estimated_remaining_minutes": 0.6761402368545533}
|
| 46 |
+
{"step": 920, "epoch": 0.4960905904556484, "timestamp": 1786748668.1569273, "loss": 8.493429565429688, "grad_norm": 1.875, "learning_rate": 0.001, "train/total_time_seconds": 373.26814346015453, "train/time_per_step_avg": 0.40516523316502573, "train/epoch_time_elapsed": 728.5175438411534, "train/estimated_remaining_minutes": 0.5409683238552964}
|
| 47 |
+
{"step": 940, "epoch": 0.5068751685090321, "timestamp": 1786748684.0965688, "loss": 8.474921417236327, "grad_norm": 1.640625, "learning_rate": 0.001, "train/total_time_seconds": 381.41837418451905, "train/time_per_step_avg": 0.40557424645870926, "train/epoch_time_elapsed": 744.4571851566434, "train/estimated_remaining_minutes": 0.40576422785587135}
|
| 48 |
+
{"step": 960, "epoch": 0.5176597465624158, "timestamp": 1786748699.8346756, "loss": 8.457220458984375, "grad_norm": 1.7890625, "learning_rate": 0.001, "train/total_time_seconds": 389.5119272880256, "train/time_per_step_avg": 0.4055132896453142, "train/epoch_time_elapsed": 760.1952919065952, "train/estimated_remaining_minutes": 0.2704943939500178}
|
| 49 |
+
{"step": 980, "epoch": 0.5284443246157994, "timestamp": 1786748715.6499183, "loss": 8.418341064453125, "grad_norm": 1.6953125, "learning_rate": 0.001, "train/total_time_seconds": 397.64077776297927, "train/time_per_step_avg": 0.4060339983180165, "train/epoch_time_elapsed": 776.0105344839394, "train/estimated_remaining_minutes": 0.13525196522550315}
|
| 50 |
+
{"step": 1000, "epoch": 0.5392289026691831, "timestamp": 1786748731.611965, "loss": 8.385391235351562, "grad_norm": 1.703125, "learning_rate": 0.001, "train/total_time_seconds": 405.79406728595495, "train/time_per_step_avg": 0.4067833938449621, "train/epoch_time_elapsed": 791.9725815802813, "train/estimated_remaining_minutes": 0.0}
|
| 51 |
+
{"step": 1000, "epoch": 0.5392289026691831, "timestamp": 1786748741.299589, "eval_loss": 2.096169948577881, "eval_runtime": 9.686, "eval_samples_per_second": 983.588, "eval_steps_per_second": 7.743, "train/total_time_seconds": 405.79406728595495, "train/time_per_step_avg": 0.4067833938449621, "train/epoch_time_elapsed": 801.6602005288005, "train/estimated_remaining_minutes": 0.0}
|
| 52 |
+
{"step": 1000, "epoch": 0.5392289026691831, "timestamp": 1786748741.3706875, "train_runtime": 802.5252, "train_samples_per_second": 637.986, "train_steps_per_second": 1.246, "total_flos": 5420315836416000.0, "train_loss": 11.859544631958007, "train/total_time_seconds": 405.79406728595495, "train/time_per_step_avg": 0.4067833938449621, "train/epoch_time_elapsed": 801.731302190572, "train/estimated_remaining_minutes": 0.0}
|
| 53 |
+
{"step": 1000, "epoch": 0.5392289026691831, "timestamp": 1786748751.1893418, "eval_loss": 2.096169948577881, "eval_runtime": 9.8163, "eval_samples_per_second": 970.533, "eval_steps_per_second": 7.64, "train/total_time_seconds": 405.79406728595495, "train/time_per_step_avg": 0.4067833938449621, "train/epoch_time_elapsed": 811.5499559640884, "train/estimated_remaining_minutes": 0.0}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/config.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"activation": "silu-waleed10",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"TinyLlamaForCausalLM"
|
| 5 |
+
],
|
| 6 |
+
"attention_bias": false,
|
| 7 |
+
"attention_dropout": 0.0,
|
| 8 |
+
"bos_token_id": 1,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"eos_token_id": 2,
|
| 11 |
+
"head_dim": 32,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 128,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 256,
|
| 16 |
+
"max_position_embeddings": 512,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"mlp_type": "mlp",
|
| 19 |
+
"model_type": "tiny_llama",
|
| 20 |
+
"num_attention_heads": 4,
|
| 21 |
+
"num_hidden_layers": 21,
|
| 22 |
+
"num_key_value_heads": 4,
|
| 23 |
+
"pad_token_id": 0,
|
| 24 |
+
"pretraining_tp": 1,
|
| 25 |
+
"rms_norm_eps": 1e-06,
|
| 26 |
+
"rope_parameters": {
|
| 27 |
+
"rope_theta": 10000.0,
|
| 28 |
+
"rope_type": "default"
|
| 29 |
+
},
|
| 30 |
+
"tie_word_embeddings": true,
|
| 31 |
+
"tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
|
| 32 |
+
"transformers_version": "5.16.0.dev0",
|
| 33 |
+
"use_cache": false,
|
| 34 |
+
"vocab_size": 4096,
|
| 35 |
+
"waleed_beta": 10.0
|
| 36 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9d9b1687dfd991ff042f68918b21a9bdfd5c2562b648e54a3cdbf2d610b3a9c9
|
| 3 |
+
size 7959296
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f2e535ffd2007a8299d146c102ef2a448eb0bc6ac87f554927192f35c0e771b6
|
| 3 |
+
size 16023738
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9d9cd6a0487226e5bd30d1846894c82af483733ab4381b75bae9c0745e05d405
|
| 3 |
+
size 14244
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:906745ab5f61e73e8c1ba850ef3b839e2766eba889b49c843836266988895165
|
| 3 |
+
size 1064
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/tokenizer_config.json
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"backend": "tokenizers",
|
| 4 |
+
"bos_token": "<|endoftext|>",
|
| 5 |
+
"eos_token": "<|endoftext|>",
|
| 6 |
+
"errors": "replace",
|
| 7 |
+
"is_local": false,
|
| 8 |
+
"local_files_only": false,
|
| 9 |
+
"model_max_length": 1024,
|
| 10 |
+
"pad_token": "<|endoftext|>",
|
| 11 |
+
"tokenizer_class": "GPT2Tokenizer",
|
| 12 |
+
"unk_token": "<|endoftext|>"
|
| 13 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/trainer_state.json
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": null,
|
| 3 |
+
"best_metric": null,
|
| 4 |
+
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.05392289026691831,
|
| 6 |
+
"eval_steps": 60000,
|
| 7 |
+
"global_step": 100,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.010784578053383662,
|
| 14 |
+
"grad_norm": 4.09375,
|
| 15 |
+
"learning_rate": 0.001,
|
| 16 |
+
"loss": 29.103460693359374,
|
| 17 |
+
"step": 20
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"epoch": 0.021569156106767323,
|
| 21 |
+
"grad_norm": 2.21875,
|
| 22 |
+
"learning_rate": 0.001,
|
| 23 |
+
"loss": 24.323658752441407,
|
| 24 |
+
"step": 40
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"epoch": 0.032353734160150985,
|
| 28 |
+
"grad_norm": 1.390625,
|
| 29 |
+
"learning_rate": 0.001,
|
| 30 |
+
"loss": 22.974208068847656,
|
| 31 |
+
"step": 60
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"epoch": 0.043138312213534646,
|
| 35 |
+
"grad_norm": 1.625,
|
| 36 |
+
"learning_rate": 0.001,
|
| 37 |
+
"loss": 21.970303344726563,
|
| 38 |
+
"step": 80
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
"epoch": 0.05392289026691831,
|
| 42 |
+
"grad_norm": 3.625,
|
| 43 |
+
"learning_rate": 0.001,
|
| 44 |
+
"loss": 20.951182556152343,
|
| 45 |
+
"step": 100
|
| 46 |
+
}
|
| 47 |
+
],
|
| 48 |
+
"logging_steps": 20,
|
| 49 |
+
"max_steps": 1000,
|
| 50 |
+
"num_input_tokens_seen": 0,
|
| 51 |
+
"num_train_epochs": 1,
|
| 52 |
+
"save_steps": 100,
|
| 53 |
+
"stateful_callbacks": {
|
| 54 |
+
"TrainerControl": {
|
| 55 |
+
"args": {
|
| 56 |
+
"should_epoch_stop": false,
|
| 57 |
+
"should_evaluate": false,
|
| 58 |
+
"should_log": false,
|
| 59 |
+
"should_save": true,
|
| 60 |
+
"should_training_stop": false
|
| 61 |
+
},
|
| 62 |
+
"attributes": {}
|
| 63 |
+
}
|
| 64 |
+
},
|
| 65 |
+
"total_flos": 542031583641600.0,
|
| 66 |
+
"train_batch_size": 128,
|
| 67 |
+
"trial_name": null,
|
| 68 |
+
"trial_params": null
|
| 69 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bda593b8a69e45e9f48a5e12fceefb418c681431f4313d1ee1004dc70bcfa0dd
|
| 3 |
+
size 4920
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/config.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"activation": "silu-waleed10",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"TinyLlamaForCausalLM"
|
| 5 |
+
],
|
| 6 |
+
"attention_bias": false,
|
| 7 |
+
"attention_dropout": 0.0,
|
| 8 |
+
"bos_token_id": 1,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"eos_token_id": 2,
|
| 11 |
+
"head_dim": 32,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 128,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 256,
|
| 16 |
+
"max_position_embeddings": 512,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"mlp_type": "mlp",
|
| 19 |
+
"model_type": "tiny_llama",
|
| 20 |
+
"num_attention_heads": 4,
|
| 21 |
+
"num_hidden_layers": 21,
|
| 22 |
+
"num_key_value_heads": 4,
|
| 23 |
+
"pad_token_id": 0,
|
| 24 |
+
"pretraining_tp": 1,
|
| 25 |
+
"rms_norm_eps": 1e-06,
|
| 26 |
+
"rope_parameters": {
|
| 27 |
+
"rope_theta": 10000.0,
|
| 28 |
+
"rope_type": "default"
|
| 29 |
+
},
|
| 30 |
+
"tie_word_embeddings": true,
|
| 31 |
+
"tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
|
| 32 |
+
"transformers_version": "5.16.0.dev0",
|
| 33 |
+
"use_cache": false,
|
| 34 |
+
"vocab_size": 4096,
|
| 35 |
+
"waleed_beta": 10.0
|
| 36 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0ba794fb6fcee5b52be8662d141115e7a75521492fb734a5fb441096c5899460
|
| 3 |
+
size 7959296
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4a30891a43b0159714e9af5b848f983099e515a8c934604658fe059bf1e5e70b
|
| 3 |
+
size 16023738
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2b66e3cc7c452b707ddac5caf0aa17618afb9bc1a0333600a22c4afb353f3165
|
| 3 |
+
size 14244
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0fc3ca8af69cc9d003a05a139e7997960ab393fe31ae1d898b29038573704b4e
|
| 3 |
+
size 1064
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/tokenizer_config.json
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"backend": "tokenizers",
|
| 4 |
+
"bos_token": "<|endoftext|>",
|
| 5 |
+
"eos_token": "<|endoftext|>",
|
| 6 |
+
"errors": "replace",
|
| 7 |
+
"is_local": false,
|
| 8 |
+
"local_files_only": false,
|
| 9 |
+
"model_max_length": 1024,
|
| 10 |
+
"pad_token": "<|endoftext|>",
|
| 11 |
+
"tokenizer_class": "GPT2Tokenizer",
|
| 12 |
+
"unk_token": "<|endoftext|>"
|
| 13 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/trainer_state.json
ADDED
|
@@ -0,0 +1,392 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": null,
|
| 3 |
+
"best_metric": null,
|
| 4 |
+
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.5392289026691831,
|
| 6 |
+
"eval_steps": 60000,
|
| 7 |
+
"global_step": 1000,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.010784578053383662,
|
| 14 |
+
"grad_norm": 4.09375,
|
| 15 |
+
"learning_rate": 0.001,
|
| 16 |
+
"loss": 29.103460693359374,
|
| 17 |
+
"step": 20
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"epoch": 0.021569156106767323,
|
| 21 |
+
"grad_norm": 2.21875,
|
| 22 |
+
"learning_rate": 0.001,
|
| 23 |
+
"loss": 24.323658752441407,
|
| 24 |
+
"step": 40
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"epoch": 0.032353734160150985,
|
| 28 |
+
"grad_norm": 1.390625,
|
| 29 |
+
"learning_rate": 0.001,
|
| 30 |
+
"loss": 22.974208068847656,
|
| 31 |
+
"step": 60
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"epoch": 0.043138312213534646,
|
| 35 |
+
"grad_norm": 1.625,
|
| 36 |
+
"learning_rate": 0.001,
|
| 37 |
+
"loss": 21.970303344726563,
|
| 38 |
+
"step": 80
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
"epoch": 0.05392289026691831,
|
| 42 |
+
"grad_norm": 3.625,
|
| 43 |
+
"learning_rate": 0.001,
|
| 44 |
+
"loss": 20.951182556152343,
|
| 45 |
+
"step": 100
|
| 46 |
+
},
|
| 47 |
+
{
|
| 48 |
+
"epoch": 0.06470746832030197,
|
| 49 |
+
"grad_norm": 3.375,
|
| 50 |
+
"learning_rate": 0.001,
|
| 51 |
+
"loss": 20.1780517578125,
|
| 52 |
+
"step": 120
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"epoch": 0.07549204637368563,
|
| 56 |
+
"grad_norm": 3.796875,
|
| 57 |
+
"learning_rate": 0.001,
|
| 58 |
+
"loss": 19.392193603515626,
|
| 59 |
+
"step": 140
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"epoch": 0.08627662442706929,
|
| 63 |
+
"grad_norm": 3.21875,
|
| 64 |
+
"learning_rate": 0.001,
|
| 65 |
+
"loss": 18.65845489501953,
|
| 66 |
+
"step": 160
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"epoch": 0.09706120248045295,
|
| 70 |
+
"grad_norm": 2.609375,
|
| 71 |
+
"learning_rate": 0.001,
|
| 72 |
+
"loss": 17.959799194335936,
|
| 73 |
+
"step": 180
|
| 74 |
+
},
|
| 75 |
+
{
|
| 76 |
+
"epoch": 0.10784578053383662,
|
| 77 |
+
"grad_norm": 2.359375,
|
| 78 |
+
"learning_rate": 0.001,
|
| 79 |
+
"loss": 17.317408752441406,
|
| 80 |
+
"step": 200
|
| 81 |
+
},
|
| 82 |
+
{
|
| 83 |
+
"epoch": 0.11863035858722028,
|
| 84 |
+
"grad_norm": 3.484375,
|
| 85 |
+
"learning_rate": 0.001,
|
| 86 |
+
"loss": 16.665341186523438,
|
| 87 |
+
"step": 220
|
| 88 |
+
},
|
| 89 |
+
{
|
| 90 |
+
"epoch": 0.12941493664060394,
|
| 91 |
+
"grad_norm": 4.0625,
|
| 92 |
+
"learning_rate": 0.001,
|
| 93 |
+
"loss": 16.151258850097655,
|
| 94 |
+
"step": 240
|
| 95 |
+
},
|
| 96 |
+
{
|
| 97 |
+
"epoch": 0.1401995146939876,
|
| 98 |
+
"grad_norm": 2.34375,
|
| 99 |
+
"learning_rate": 0.001,
|
| 100 |
+
"loss": 15.6710205078125,
|
| 101 |
+
"step": 260
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"epoch": 0.15098409274737126,
|
| 105 |
+
"grad_norm": 2.984375,
|
| 106 |
+
"learning_rate": 0.001,
|
| 107 |
+
"loss": 15.264056396484374,
|
| 108 |
+
"step": 280
|
| 109 |
+
},
|
| 110 |
+
{
|
| 111 |
+
"epoch": 0.1617686708007549,
|
| 112 |
+
"grad_norm": 1.8125,
|
| 113 |
+
"learning_rate": 0.001,
|
| 114 |
+
"loss": 14.908853149414062,
|
| 115 |
+
"step": 300
|
| 116 |
+
},
|
| 117 |
+
{
|
| 118 |
+
"epoch": 0.17255324885413859,
|
| 119 |
+
"grad_norm": 2.390625,
|
| 120 |
+
"learning_rate": 0.001,
|
| 121 |
+
"loss": 14.611824035644531,
|
| 122 |
+
"step": 320
|
| 123 |
+
},
|
| 124 |
+
{
|
| 125 |
+
"epoch": 0.18333782690752223,
|
| 126 |
+
"grad_norm": 1.6484375,
|
| 127 |
+
"learning_rate": 0.001,
|
| 128 |
+
"loss": 14.270545959472656,
|
| 129 |
+
"step": 340
|
| 130 |
+
},
|
| 131 |
+
{
|
| 132 |
+
"epoch": 0.1941224049609059,
|
| 133 |
+
"grad_norm": 2.03125,
|
| 134 |
+
"learning_rate": 0.001,
|
| 135 |
+
"loss": 13.935658264160157,
|
| 136 |
+
"step": 360
|
| 137 |
+
},
|
| 138 |
+
{
|
| 139 |
+
"epoch": 0.20490698301428956,
|
| 140 |
+
"grad_norm": 2.75,
|
| 141 |
+
"learning_rate": 0.001,
|
| 142 |
+
"loss": 13.626214599609375,
|
| 143 |
+
"step": 380
|
| 144 |
+
},
|
| 145 |
+
{
|
| 146 |
+
"epoch": 0.21569156106767323,
|
| 147 |
+
"grad_norm": 2.6875,
|
| 148 |
+
"learning_rate": 0.001,
|
| 149 |
+
"loss": 13.423133850097656,
|
| 150 |
+
"step": 400
|
| 151 |
+
},
|
| 152 |
+
{
|
| 153 |
+
"epoch": 0.22647613912105688,
|
| 154 |
+
"grad_norm": 2.640625,
|
| 155 |
+
"learning_rate": 0.001,
|
| 156 |
+
"loss": 13.1606201171875,
|
| 157 |
+
"step": 420
|
| 158 |
+
},
|
| 159 |
+
{
|
| 160 |
+
"epoch": 0.23726071717444056,
|
| 161 |
+
"grad_norm": 2.03125,
|
| 162 |
+
"learning_rate": 0.001,
|
| 163 |
+
"loss": 12.97046356201172,
|
| 164 |
+
"step": 440
|
| 165 |
+
},
|
| 166 |
+
{
|
| 167 |
+
"epoch": 0.2480452952278242,
|
| 168 |
+
"grad_norm": 2.375,
|
| 169 |
+
"learning_rate": 0.001,
|
| 170 |
+
"loss": 12.767843627929688,
|
| 171 |
+
"step": 460
|
| 172 |
+
},
|
| 173 |
+
{
|
| 174 |
+
"epoch": 0.2588298732812079,
|
| 175 |
+
"grad_norm": 1.9609375,
|
| 176 |
+
"learning_rate": 0.001,
|
| 177 |
+
"loss": 12.555582427978516,
|
| 178 |
+
"step": 480
|
| 179 |
+
},
|
| 180 |
+
{
|
| 181 |
+
"epoch": 0.26961445133459155,
|
| 182 |
+
"grad_norm": 1.6640625,
|
| 183 |
+
"learning_rate": 0.001,
|
| 184 |
+
"loss": 12.363282012939454,
|
| 185 |
+
"step": 500
|
| 186 |
+
},
|
| 187 |
+
{
|
| 188 |
+
"epoch": 0.2803990293879752,
|
| 189 |
+
"grad_norm": 2.40625,
|
| 190 |
+
"learning_rate": 0.001,
|
| 191 |
+
"loss": 12.229927825927735,
|
| 192 |
+
"step": 520
|
| 193 |
+
},
|
| 194 |
+
{
|
| 195 |
+
"epoch": 0.29118360744135885,
|
| 196 |
+
"grad_norm": 2.1875,
|
| 197 |
+
"learning_rate": 0.001,
|
| 198 |
+
"loss": 12.06595230102539,
|
| 199 |
+
"step": 540
|
| 200 |
+
},
|
| 201 |
+
{
|
| 202 |
+
"epoch": 0.3019681854947425,
|
| 203 |
+
"grad_norm": 2.25,
|
| 204 |
+
"learning_rate": 0.001,
|
| 205 |
+
"loss": 11.916503143310546,
|
| 206 |
+
"step": 560
|
| 207 |
+
},
|
| 208 |
+
{
|
| 209 |
+
"epoch": 0.3127527635481262,
|
| 210 |
+
"grad_norm": 2.0625,
|
| 211 |
+
"learning_rate": 0.001,
|
| 212 |
+
"loss": 11.778754425048827,
|
| 213 |
+
"step": 580
|
| 214 |
+
},
|
| 215 |
+
{
|
| 216 |
+
"epoch": 0.3235373416015098,
|
| 217 |
+
"grad_norm": 2.5625,
|
| 218 |
+
"learning_rate": 0.001,
|
| 219 |
+
"loss": 11.674578857421874,
|
| 220 |
+
"step": 600
|
| 221 |
+
},
|
| 222 |
+
{
|
| 223 |
+
"epoch": 0.3343219196548935,
|
| 224 |
+
"grad_norm": 1.9921875,
|
| 225 |
+
"learning_rate": 0.001,
|
| 226 |
+
"loss": 11.54565658569336,
|
| 227 |
+
"step": 620
|
| 228 |
+
},
|
| 229 |
+
{
|
| 230 |
+
"epoch": 0.34510649770827717,
|
| 231 |
+
"grad_norm": 2.0,
|
| 232 |
+
"learning_rate": 0.001,
|
| 233 |
+
"loss": 11.434559631347657,
|
| 234 |
+
"step": 640
|
| 235 |
+
},
|
| 236 |
+
{
|
| 237 |
+
"epoch": 0.35589107576166085,
|
| 238 |
+
"grad_norm": 2.203125,
|
| 239 |
+
"learning_rate": 0.001,
|
| 240 |
+
"loss": 11.319635772705078,
|
| 241 |
+
"step": 660
|
| 242 |
+
},
|
| 243 |
+
{
|
| 244 |
+
"epoch": 0.36667565381504447,
|
| 245 |
+
"grad_norm": 2.15625,
|
| 246 |
+
"learning_rate": 0.001,
|
| 247 |
+
"loss": 11.202098846435547,
|
| 248 |
+
"step": 680
|
| 249 |
+
},
|
| 250 |
+
{
|
| 251 |
+
"epoch": 0.37746023186842814,
|
| 252 |
+
"grad_norm": 2.421875,
|
| 253 |
+
"learning_rate": 0.001,
|
| 254 |
+
"loss": 11.103318023681641,
|
| 255 |
+
"step": 700
|
| 256 |
+
},
|
| 257 |
+
{
|
| 258 |
+
"epoch": 0.3882448099218118,
|
| 259 |
+
"grad_norm": 2.03125,
|
| 260 |
+
"learning_rate": 0.001,
|
| 261 |
+
"loss": 10.984959411621094,
|
| 262 |
+
"step": 720
|
| 263 |
+
},
|
| 264 |
+
{
|
| 265 |
+
"epoch": 0.3990293879751955,
|
| 266 |
+
"grad_norm": 2.78125,
|
| 267 |
+
"learning_rate": 0.001,
|
| 268 |
+
"loss": 10.932718658447266,
|
| 269 |
+
"step": 740
|
| 270 |
+
},
|
| 271 |
+
{
|
| 272 |
+
"epoch": 0.4098139660285791,
|
| 273 |
+
"grad_norm": 2.265625,
|
| 274 |
+
"learning_rate": 0.001,
|
| 275 |
+
"loss": 10.841696166992188,
|
| 276 |
+
"step": 760
|
| 277 |
+
},
|
| 278 |
+
{
|
| 279 |
+
"epoch": 0.4205985440819628,
|
| 280 |
+
"grad_norm": 2.53125,
|
| 281 |
+
"learning_rate": 0.001,
|
| 282 |
+
"loss": 10.763890838623047,
|
| 283 |
+
"step": 780
|
| 284 |
+
},
|
| 285 |
+
{
|
| 286 |
+
"epoch": 0.43138312213534646,
|
| 287 |
+
"grad_norm": 2.375,
|
| 288 |
+
"learning_rate": 0.001,
|
| 289 |
+
"loss": 10.633779907226563,
|
| 290 |
+
"step": 800
|
| 291 |
+
},
|
| 292 |
+
{
|
| 293 |
+
"epoch": 0.44216770018873014,
|
| 294 |
+
"grad_norm": 2.46875,
|
| 295 |
+
"learning_rate": 0.001,
|
| 296 |
+
"loss": 10.587508392333984,
|
| 297 |
+
"step": 820
|
| 298 |
+
},
|
| 299 |
+
{
|
| 300 |
+
"epoch": 0.45295227824211376,
|
| 301 |
+
"grad_norm": 2.53125,
|
| 302 |
+
"learning_rate": 0.001,
|
| 303 |
+
"loss": 10.518219757080079,
|
| 304 |
+
"step": 840
|
| 305 |
+
},
|
| 306 |
+
{
|
| 307 |
+
"epoch": 0.46373685629549743,
|
| 308 |
+
"grad_norm": 2.484375,
|
| 309 |
+
"learning_rate": 0.001,
|
| 310 |
+
"loss": 10.428038787841796,
|
| 311 |
+
"step": 860
|
| 312 |
+
},
|
| 313 |
+
{
|
| 314 |
+
"epoch": 0.4745214343488811,
|
| 315 |
+
"grad_norm": 2.234375,
|
| 316 |
+
"learning_rate": 0.001,
|
| 317 |
+
"loss": 10.359873199462891,
|
| 318 |
+
"step": 880
|
| 319 |
+
},
|
| 320 |
+
{
|
| 321 |
+
"epoch": 0.4853060124022648,
|
| 322 |
+
"grad_norm": 2.640625,
|
| 323 |
+
"learning_rate": 0.001,
|
| 324 |
+
"loss": 10.303240966796874,
|
| 325 |
+
"step": 900
|
| 326 |
+
},
|
| 327 |
+
{
|
| 328 |
+
"epoch": 0.4960905904556484,
|
| 329 |
+
"grad_norm": 2.703125,
|
| 330 |
+
"learning_rate": 0.001,
|
| 331 |
+
"loss": 10.233261108398438,
|
| 332 |
+
"step": 920
|
| 333 |
+
},
|
| 334 |
+
{
|
| 335 |
+
"epoch": 0.5068751685090321,
|
| 336 |
+
"grad_norm": 2.984375,
|
| 337 |
+
"learning_rate": 0.001,
|
| 338 |
+
"loss": 10.198040008544922,
|
| 339 |
+
"step": 940
|
| 340 |
+
},
|
| 341 |
+
{
|
| 342 |
+
"epoch": 0.5176597465624158,
|
| 343 |
+
"grad_norm": 2.234375,
|
| 344 |
+
"learning_rate": 0.001,
|
| 345 |
+
"loss": 10.153294372558594,
|
| 346 |
+
"step": 960
|
| 347 |
+
},
|
| 348 |
+
{
|
| 349 |
+
"epoch": 0.5284443246157994,
|
| 350 |
+
"grad_norm": 2.46875,
|
| 351 |
+
"learning_rate": 0.001,
|
| 352 |
+
"loss": 10.112586212158202,
|
| 353 |
+
"step": 980
|
| 354 |
+
},
|
| 355 |
+
{
|
| 356 |
+
"epoch": 0.5392289026691831,
|
| 357 |
+
"grad_norm": 2.609375,
|
| 358 |
+
"learning_rate": 0.001,
|
| 359 |
+
"loss": 10.053606414794922,
|
| 360 |
+
"step": 1000
|
| 361 |
+
},
|
| 362 |
+
{
|
| 363 |
+
"epoch": 0.5392289026691831,
|
| 364 |
+
"eval_loss": 2.5052456855773926,
|
| 365 |
+
"eval_runtime": 9.4135,
|
| 366 |
+
"eval_samples_per_second": 1012.06,
|
| 367 |
+
"eval_steps_per_second": 7.967,
|
| 368 |
+
"step": 1000
|
| 369 |
+
}
|
| 370 |
+
],
|
| 371 |
+
"logging_steps": 20,
|
| 372 |
+
"max_steps": 1000,
|
| 373 |
+
"num_input_tokens_seen": 0,
|
| 374 |
+
"num_train_epochs": 1,
|
| 375 |
+
"save_steps": 100,
|
| 376 |
+
"stateful_callbacks": {
|
| 377 |
+
"TrainerControl": {
|
| 378 |
+
"args": {
|
| 379 |
+
"should_epoch_stop": false,
|
| 380 |
+
"should_evaluate": false,
|
| 381 |
+
"should_log": false,
|
| 382 |
+
"should_save": true,
|
| 383 |
+
"should_training_stop": true
|
| 384 |
+
},
|
| 385 |
+
"attributes": {}
|
| 386 |
+
}
|
| 387 |
+
},
|
| 388 |
+
"total_flos": 5420315836416000.0,
|
| 389 |
+
"train_batch_size": 128,
|
| 390 |
+
"trial_name": null,
|
| 391 |
+
"trial_params": null
|
| 392 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bda593b8a69e45e9f48a5e12fceefb418c681431f4313d1ee1004dc70bcfa0dd
|
| 3 |
+
size 4920
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/config.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"activation": "silu-waleed10",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"TinyLlamaForCausalLM"
|
| 5 |
+
],
|
| 6 |
+
"attention_bias": false,
|
| 7 |
+
"attention_dropout": 0.0,
|
| 8 |
+
"bos_token_id": 1,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"eos_token_id": 2,
|
| 11 |
+
"head_dim": 32,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 128,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 256,
|
| 16 |
+
"max_position_embeddings": 512,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"mlp_type": "mlp",
|
| 19 |
+
"model_type": "tiny_llama",
|
| 20 |
+
"num_attention_heads": 4,
|
| 21 |
+
"num_hidden_layers": 21,
|
| 22 |
+
"num_key_value_heads": 4,
|
| 23 |
+
"pad_token_id": 0,
|
| 24 |
+
"pretraining_tp": 1,
|
| 25 |
+
"rms_norm_eps": 1e-06,
|
| 26 |
+
"rope_parameters": {
|
| 27 |
+
"rope_theta": 10000.0,
|
| 28 |
+
"rope_type": "default"
|
| 29 |
+
},
|
| 30 |
+
"tie_word_embeddings": true,
|
| 31 |
+
"tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
|
| 32 |
+
"transformers_version": "5.16.0.dev0",
|
| 33 |
+
"use_cache": false,
|
| 34 |
+
"vocab_size": 4096,
|
| 35 |
+
"waleed_beta": 10.0
|
| 36 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e47052a6a249ebf53373e960f46db44c16badf55d0b1f5641d7b6fe7bcd56add
|
| 3 |
+
size 7959296
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a689448d286470826d3e90b92a69afbf00fcf27b42685e171aceb475c52d12bc
|
| 3 |
+
size 16023738
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9d9cd6a0487226e5bd30d1846894c82af483733ab4381b75bae9c0745e05d405
|
| 3 |
+
size 14244
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2d3b5a970fc92af42664e84ba686059eaaacfaa668691c0cf6d2afddda677461
|
| 3 |
+
size 1064
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/tokenizer_config.json
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"backend": "tokenizers",
|
| 4 |
+
"bos_token": "<|endoftext|>",
|
| 5 |
+
"eos_token": "<|endoftext|>",
|
| 6 |
+
"errors": "replace",
|
| 7 |
+
"is_local": false,
|
| 8 |
+
"local_files_only": false,
|
| 9 |
+
"model_max_length": 1024,
|
| 10 |
+
"pad_token": "<|endoftext|>",
|
| 11 |
+
"tokenizer_class": "GPT2Tokenizer",
|
| 12 |
+
"unk_token": "<|endoftext|>"
|
| 13 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/trainer_state.json
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": null,
|
| 3 |
+
"best_metric": null,
|
| 4 |
+
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.10784578053383662,
|
| 6 |
+
"eval_steps": 60000,
|
| 7 |
+
"global_step": 200,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.010784578053383662,
|
| 14 |
+
"grad_norm": 4.09375,
|
| 15 |
+
"learning_rate": 0.001,
|
| 16 |
+
"loss": 29.103460693359374,
|
| 17 |
+
"step": 20
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"epoch": 0.021569156106767323,
|
| 21 |
+
"grad_norm": 2.21875,
|
| 22 |
+
"learning_rate": 0.001,
|
| 23 |
+
"loss": 24.323658752441407,
|
| 24 |
+
"step": 40
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"epoch": 0.032353734160150985,
|
| 28 |
+
"grad_norm": 1.390625,
|
| 29 |
+
"learning_rate": 0.001,
|
| 30 |
+
"loss": 22.974208068847656,
|
| 31 |
+
"step": 60
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"epoch": 0.043138312213534646,
|
| 35 |
+
"grad_norm": 1.625,
|
| 36 |
+
"learning_rate": 0.001,
|
| 37 |
+
"loss": 21.970303344726563,
|
| 38 |
+
"step": 80
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
"epoch": 0.05392289026691831,
|
| 42 |
+
"grad_norm": 3.625,
|
| 43 |
+
"learning_rate": 0.001,
|
| 44 |
+
"loss": 20.951182556152343,
|
| 45 |
+
"step": 100
|
| 46 |
+
},
|
| 47 |
+
{
|
| 48 |
+
"epoch": 0.06470746832030197,
|
| 49 |
+
"grad_norm": 3.375,
|
| 50 |
+
"learning_rate": 0.001,
|
| 51 |
+
"loss": 20.1780517578125,
|
| 52 |
+
"step": 120
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"epoch": 0.07549204637368563,
|
| 56 |
+
"grad_norm": 3.796875,
|
| 57 |
+
"learning_rate": 0.001,
|
| 58 |
+
"loss": 19.392193603515626,
|
| 59 |
+
"step": 140
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"epoch": 0.08627662442706929,
|
| 63 |
+
"grad_norm": 3.21875,
|
| 64 |
+
"learning_rate": 0.001,
|
| 65 |
+
"loss": 18.65845489501953,
|
| 66 |
+
"step": 160
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"epoch": 0.09706120248045295,
|
| 70 |
+
"grad_norm": 2.609375,
|
| 71 |
+
"learning_rate": 0.001,
|
| 72 |
+
"loss": 17.959799194335936,
|
| 73 |
+
"step": 180
|
| 74 |
+
},
|
| 75 |
+
{
|
| 76 |
+
"epoch": 0.10784578053383662,
|
| 77 |
+
"grad_norm": 2.359375,
|
| 78 |
+
"learning_rate": 0.001,
|
| 79 |
+
"loss": 17.317408752441406,
|
| 80 |
+
"step": 200
|
| 81 |
+
}
|
| 82 |
+
],
|
| 83 |
+
"logging_steps": 20,
|
| 84 |
+
"max_steps": 1000,
|
| 85 |
+
"num_input_tokens_seen": 0,
|
| 86 |
+
"num_train_epochs": 1,
|
| 87 |
+
"save_steps": 100,
|
| 88 |
+
"stateful_callbacks": {
|
| 89 |
+
"TrainerControl": {
|
| 90 |
+
"args": {
|
| 91 |
+
"should_epoch_stop": false,
|
| 92 |
+
"should_evaluate": false,
|
| 93 |
+
"should_log": false,
|
| 94 |
+
"should_save": true,
|
| 95 |
+
"should_training_stop": false
|
| 96 |
+
},
|
| 97 |
+
"attributes": {}
|
| 98 |
+
}
|
| 99 |
+
},
|
| 100 |
+
"total_flos": 1084063167283200.0,
|
| 101 |
+
"train_batch_size": 128,
|
| 102 |
+
"trial_name": null,
|
| 103 |
+
"trial_params": null
|
| 104 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bda593b8a69e45e9f48a5e12fceefb418c681431f4313d1ee1004dc70bcfa0dd
|
| 3 |
+
size 4920
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/config.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"activation": "silu-waleed10",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"TinyLlamaForCausalLM"
|
| 5 |
+
],
|
| 6 |
+
"attention_bias": false,
|
| 7 |
+
"attention_dropout": 0.0,
|
| 8 |
+
"bos_token_id": 1,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"eos_token_id": 2,
|
| 11 |
+
"head_dim": 32,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 128,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 256,
|
| 16 |
+
"max_position_embeddings": 512,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"mlp_type": "mlp",
|
| 19 |
+
"model_type": "tiny_llama",
|
| 20 |
+
"num_attention_heads": 4,
|
| 21 |
+
"num_hidden_layers": 21,
|
| 22 |
+
"num_key_value_heads": 4,
|
| 23 |
+
"pad_token_id": 0,
|
| 24 |
+
"pretraining_tp": 1,
|
| 25 |
+
"rms_norm_eps": 1e-06,
|
| 26 |
+
"rope_parameters": {
|
| 27 |
+
"rope_theta": 10000.0,
|
| 28 |
+
"rope_type": "default"
|
| 29 |
+
},
|
| 30 |
+
"tie_word_embeddings": true,
|
| 31 |
+
"tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
|
| 32 |
+
"transformers_version": "5.16.0.dev0",
|
| 33 |
+
"use_cache": false,
|
| 34 |
+
"vocab_size": 4096,
|
| 35 |
+
"waleed_beta": 10.0
|
| 36 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f515f27d365a121e29deb1fb378e30dd334e121d4cb62d906d24656b71f2113e
|
| 3 |
+
size 7959296
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:326ff18127cbac66b90e9041ed245dfd03d7ad9b8654dc096a075bfe445183a6
|
| 3 |
+
size 16023738
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9d9cd6a0487226e5bd30d1846894c82af483733ab4381b75bae9c0745e05d405
|
| 3 |
+
size 14244
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:76cd21f32221b90fb8288fd2428595ec8b5039fe2425086d67cacb5c8d814c9e
|
| 3 |
+
size 1064
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/tokenizer_config.json
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"backend": "tokenizers",
|
| 4 |
+
"bos_token": "<|endoftext|>",
|
| 5 |
+
"eos_token": "<|endoftext|>",
|
| 6 |
+
"errors": "replace",
|
| 7 |
+
"is_local": false,
|
| 8 |
+
"local_files_only": false,
|
| 9 |
+
"model_max_length": 1024,
|
| 10 |
+
"pad_token": "<|endoftext|>",
|
| 11 |
+
"tokenizer_class": "GPT2Tokenizer",
|
| 12 |
+
"unk_token": "<|endoftext|>"
|
| 13 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/trainer_state.json
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": null,
|
| 3 |
+
"best_metric": null,
|
| 4 |
+
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.1617686708007549,
|
| 6 |
+
"eval_steps": 60000,
|
| 7 |
+
"global_step": 300,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.010784578053383662,
|
| 14 |
+
"grad_norm": 4.09375,
|
| 15 |
+
"learning_rate": 0.001,
|
| 16 |
+
"loss": 29.103460693359374,
|
| 17 |
+
"step": 20
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"epoch": 0.021569156106767323,
|
| 21 |
+
"grad_norm": 2.21875,
|
| 22 |
+
"learning_rate": 0.001,
|
| 23 |
+
"loss": 24.323658752441407,
|
| 24 |
+
"step": 40
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"epoch": 0.032353734160150985,
|
| 28 |
+
"grad_norm": 1.390625,
|
| 29 |
+
"learning_rate": 0.001,
|
| 30 |
+
"loss": 22.974208068847656,
|
| 31 |
+
"step": 60
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"epoch": 0.043138312213534646,
|
| 35 |
+
"grad_norm": 1.625,
|
| 36 |
+
"learning_rate": 0.001,
|
| 37 |
+
"loss": 21.970303344726563,
|
| 38 |
+
"step": 80
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
"epoch": 0.05392289026691831,
|
| 42 |
+
"grad_norm": 3.625,
|
| 43 |
+
"learning_rate": 0.001,
|
| 44 |
+
"loss": 20.951182556152343,
|
| 45 |
+
"step": 100
|
| 46 |
+
},
|
| 47 |
+
{
|
| 48 |
+
"epoch": 0.06470746832030197,
|
| 49 |
+
"grad_norm": 3.375,
|
| 50 |
+
"learning_rate": 0.001,
|
| 51 |
+
"loss": 20.1780517578125,
|
| 52 |
+
"step": 120
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"epoch": 0.07549204637368563,
|
| 56 |
+
"grad_norm": 3.796875,
|
| 57 |
+
"learning_rate": 0.001,
|
| 58 |
+
"loss": 19.392193603515626,
|
| 59 |
+
"step": 140
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"epoch": 0.08627662442706929,
|
| 63 |
+
"grad_norm": 3.21875,
|
| 64 |
+
"learning_rate": 0.001,
|
| 65 |
+
"loss": 18.65845489501953,
|
| 66 |
+
"step": 160
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"epoch": 0.09706120248045295,
|
| 70 |
+
"grad_norm": 2.609375,
|
| 71 |
+
"learning_rate": 0.001,
|
| 72 |
+
"loss": 17.959799194335936,
|
| 73 |
+
"step": 180
|
| 74 |
+
},
|
| 75 |
+
{
|
| 76 |
+
"epoch": 0.10784578053383662,
|
| 77 |
+
"grad_norm": 2.359375,
|
| 78 |
+
"learning_rate": 0.001,
|
| 79 |
+
"loss": 17.317408752441406,
|
| 80 |
+
"step": 200
|
| 81 |
+
},
|
| 82 |
+
{
|
| 83 |
+
"epoch": 0.11863035858722028,
|
| 84 |
+
"grad_norm": 3.484375,
|
| 85 |
+
"learning_rate": 0.001,
|
| 86 |
+
"loss": 16.665341186523438,
|
| 87 |
+
"step": 220
|
| 88 |
+
},
|
| 89 |
+
{
|
| 90 |
+
"epoch": 0.12941493664060394,
|
| 91 |
+
"grad_norm": 4.0625,
|
| 92 |
+
"learning_rate": 0.001,
|
| 93 |
+
"loss": 16.151258850097655,
|
| 94 |
+
"step": 240
|
| 95 |
+
},
|
| 96 |
+
{
|
| 97 |
+
"epoch": 0.1401995146939876,
|
| 98 |
+
"grad_norm": 2.34375,
|
| 99 |
+
"learning_rate": 0.001,
|
| 100 |
+
"loss": 15.6710205078125,
|
| 101 |
+
"step": 260
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"epoch": 0.15098409274737126,
|
| 105 |
+
"grad_norm": 2.984375,
|
| 106 |
+
"learning_rate": 0.001,
|
| 107 |
+
"loss": 15.264056396484374,
|
| 108 |
+
"step": 280
|
| 109 |
+
},
|
| 110 |
+
{
|
| 111 |
+
"epoch": 0.1617686708007549,
|
| 112 |
+
"grad_norm": 1.8125,
|
| 113 |
+
"learning_rate": 0.001,
|
| 114 |
+
"loss": 14.908853149414062,
|
| 115 |
+
"step": 300
|
| 116 |
+
}
|
| 117 |
+
],
|
| 118 |
+
"logging_steps": 20,
|
| 119 |
+
"max_steps": 1000,
|
| 120 |
+
"num_input_tokens_seen": 0,
|
| 121 |
+
"num_train_epochs": 1,
|
| 122 |
+
"save_steps": 100,
|
| 123 |
+
"stateful_callbacks": {
|
| 124 |
+
"TrainerControl": {
|
| 125 |
+
"args": {
|
| 126 |
+
"should_epoch_stop": false,
|
| 127 |
+
"should_evaluate": false,
|
| 128 |
+
"should_log": false,
|
| 129 |
+
"should_save": true,
|
| 130 |
+
"should_training_stop": false
|
| 131 |
+
},
|
| 132 |
+
"attributes": {}
|
| 133 |
+
}
|
| 134 |
+
},
|
| 135 |
+
"total_flos": 1626094750924800.0,
|
| 136 |
+
"train_batch_size": 128,
|
| 137 |
+
"trial_name": null,
|
| 138 |
+
"trial_params": null
|
| 139 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bda593b8a69e45e9f48a5e12fceefb418c681431f4313d1ee1004dc70bcfa0dd
|
| 3 |
+
size 4920
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/config.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"activation": "silu-waleed10",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"TinyLlamaForCausalLM"
|
| 5 |
+
],
|
| 6 |
+
"attention_bias": false,
|
| 7 |
+
"attention_dropout": 0.0,
|
| 8 |
+
"bos_token_id": 1,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"eos_token_id": 2,
|
| 11 |
+
"head_dim": 32,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 128,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 256,
|
| 16 |
+
"max_position_embeddings": 512,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"mlp_type": "mlp",
|
| 19 |
+
"model_type": "tiny_llama",
|
| 20 |
+
"num_attention_heads": 4,
|
| 21 |
+
"num_hidden_layers": 21,
|
| 22 |
+
"num_key_value_heads": 4,
|
| 23 |
+
"pad_token_id": 0,
|
| 24 |
+
"pretraining_tp": 1,
|
| 25 |
+
"rms_norm_eps": 1e-06,
|
| 26 |
+
"rope_parameters": {
|
| 27 |
+
"rope_theta": 10000.0,
|
| 28 |
+
"rope_type": "default"
|
| 29 |
+
},
|
| 30 |
+
"tie_word_embeddings": true,
|
| 31 |
+
"tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
|
| 32 |
+
"transformers_version": "5.16.0.dev0",
|
| 33 |
+
"use_cache": false,
|
| 34 |
+
"vocab_size": 4096,
|
| 35 |
+
"waleed_beta": 10.0
|
| 36 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8a38c2c8a915b9b6f40c28c42f072a644b4cf5db90c25784335b26df0418b09b
|
| 3 |
+
size 7959296
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4855eab9d35bda1625a50a9b1b82dd1490b389965177b34af7bab3d07f5e0396
|
| 3 |
+
size 16023738
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9d9cd6a0487226e5bd30d1846894c82af483733ab4381b75bae9c0745e05d405
|
| 3 |
+
size 14244
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:51769a11c87b2b665cbe64c58b934afdb1fa1998bffb4addcc2f852b172d6681
|
| 3 |
+
size 1064
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/tokenizer_config.json
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"backend": "tokenizers",
|
| 4 |
+
"bos_token": "<|endoftext|>",
|
| 5 |
+
"eos_token": "<|endoftext|>",
|
| 6 |
+
"errors": "replace",
|
| 7 |
+
"is_local": false,
|
| 8 |
+
"local_files_only": false,
|
| 9 |
+
"model_max_length": 1024,
|
| 10 |
+
"pad_token": "<|endoftext|>",
|
| 11 |
+
"tokenizer_class": "GPT2Tokenizer",
|
| 12 |
+
"unk_token": "<|endoftext|>"
|
| 13 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/trainer_state.json
ADDED
|
@@ -0,0 +1,174 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": null,
|
| 3 |
+
"best_metric": null,
|
| 4 |
+
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.21569156106767323,
|
| 6 |
+
"eval_steps": 60000,
|
| 7 |
+
"global_step": 400,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.010784578053383662,
|
| 14 |
+
"grad_norm": 4.09375,
|
| 15 |
+
"learning_rate": 0.001,
|
| 16 |
+
"loss": 29.103460693359374,
|
| 17 |
+
"step": 20
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"epoch": 0.021569156106767323,
|
| 21 |
+
"grad_norm": 2.21875,
|
| 22 |
+
"learning_rate": 0.001,
|
| 23 |
+
"loss": 24.323658752441407,
|
| 24 |
+
"step": 40
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"epoch": 0.032353734160150985,
|
| 28 |
+
"grad_norm": 1.390625,
|
| 29 |
+
"learning_rate": 0.001,
|
| 30 |
+
"loss": 22.974208068847656,
|
| 31 |
+
"step": 60
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"epoch": 0.043138312213534646,
|
| 35 |
+
"grad_norm": 1.625,
|
| 36 |
+
"learning_rate": 0.001,
|
| 37 |
+
"loss": 21.970303344726563,
|
| 38 |
+
"step": 80
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
"epoch": 0.05392289026691831,
|
| 42 |
+
"grad_norm": 3.625,
|
| 43 |
+
"learning_rate": 0.001,
|
| 44 |
+
"loss": 20.951182556152343,
|
| 45 |
+
"step": 100
|
| 46 |
+
},
|
| 47 |
+
{
|
| 48 |
+
"epoch": 0.06470746832030197,
|
| 49 |
+
"grad_norm": 3.375,
|
| 50 |
+
"learning_rate": 0.001,
|
| 51 |
+
"loss": 20.1780517578125,
|
| 52 |
+
"step": 120
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"epoch": 0.07549204637368563,
|
| 56 |
+
"grad_norm": 3.796875,
|
| 57 |
+
"learning_rate": 0.001,
|
| 58 |
+
"loss": 19.392193603515626,
|
| 59 |
+
"step": 140
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"epoch": 0.08627662442706929,
|
| 63 |
+
"grad_norm": 3.21875,
|
| 64 |
+
"learning_rate": 0.001,
|
| 65 |
+
"loss": 18.65845489501953,
|
| 66 |
+
"step": 160
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"epoch": 0.09706120248045295,
|
| 70 |
+
"grad_norm": 2.609375,
|
| 71 |
+
"learning_rate": 0.001,
|
| 72 |
+
"loss": 17.959799194335936,
|
| 73 |
+
"step": 180
|
| 74 |
+
},
|
| 75 |
+
{
|
| 76 |
+
"epoch": 0.10784578053383662,
|
| 77 |
+
"grad_norm": 2.359375,
|
| 78 |
+
"learning_rate": 0.001,
|
| 79 |
+
"loss": 17.317408752441406,
|
| 80 |
+
"step": 200
|
| 81 |
+
},
|
| 82 |
+
{
|
| 83 |
+
"epoch": 0.11863035858722028,
|
| 84 |
+
"grad_norm": 3.484375,
|
| 85 |
+
"learning_rate": 0.001,
|
| 86 |
+
"loss": 16.665341186523438,
|
| 87 |
+
"step": 220
|
| 88 |
+
},
|
| 89 |
+
{
|
| 90 |
+
"epoch": 0.12941493664060394,
|
| 91 |
+
"grad_norm": 4.0625,
|
| 92 |
+
"learning_rate": 0.001,
|
| 93 |
+
"loss": 16.151258850097655,
|
| 94 |
+
"step": 240
|
| 95 |
+
},
|
| 96 |
+
{
|
| 97 |
+
"epoch": 0.1401995146939876,
|
| 98 |
+
"grad_norm": 2.34375,
|
| 99 |
+
"learning_rate": 0.001,
|
| 100 |
+
"loss": 15.6710205078125,
|
| 101 |
+
"step": 260
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"epoch": 0.15098409274737126,
|
| 105 |
+
"grad_norm": 2.984375,
|
| 106 |
+
"learning_rate": 0.001,
|
| 107 |
+
"loss": 15.264056396484374,
|
| 108 |
+
"step": 280
|
| 109 |
+
},
|
| 110 |
+
{
|
| 111 |
+
"epoch": 0.1617686708007549,
|
| 112 |
+
"grad_norm": 1.8125,
|
| 113 |
+
"learning_rate": 0.001,
|
| 114 |
+
"loss": 14.908853149414062,
|
| 115 |
+
"step": 300
|
| 116 |
+
},
|
| 117 |
+
{
|
| 118 |
+
"epoch": 0.17255324885413859,
|
| 119 |
+
"grad_norm": 2.390625,
|
| 120 |
+
"learning_rate": 0.001,
|
| 121 |
+
"loss": 14.611824035644531,
|
| 122 |
+
"step": 320
|
| 123 |
+
},
|
| 124 |
+
{
|
| 125 |
+
"epoch": 0.18333782690752223,
|
| 126 |
+
"grad_norm": 1.6484375,
|
| 127 |
+
"learning_rate": 0.001,
|
| 128 |
+
"loss": 14.270545959472656,
|
| 129 |
+
"step": 340
|
| 130 |
+
},
|
| 131 |
+
{
|
| 132 |
+
"epoch": 0.1941224049609059,
|
| 133 |
+
"grad_norm": 2.03125,
|
| 134 |
+
"learning_rate": 0.001,
|
| 135 |
+
"loss": 13.935658264160157,
|
| 136 |
+
"step": 360
|
| 137 |
+
},
|
| 138 |
+
{
|
| 139 |
+
"epoch": 0.20490698301428956,
|
| 140 |
+
"grad_norm": 2.75,
|
| 141 |
+
"learning_rate": 0.001,
|
| 142 |
+
"loss": 13.626214599609375,
|
| 143 |
+
"step": 380
|
| 144 |
+
},
|
| 145 |
+
{
|
| 146 |
+
"epoch": 0.21569156106767323,
|
| 147 |
+
"grad_norm": 2.6875,
|
| 148 |
+
"learning_rate": 0.001,
|
| 149 |
+
"loss": 13.423133850097656,
|
| 150 |
+
"step": 400
|
| 151 |
+
}
|
| 152 |
+
],
|
| 153 |
+
"logging_steps": 20,
|
| 154 |
+
"max_steps": 1000,
|
| 155 |
+
"num_input_tokens_seen": 0,
|
| 156 |
+
"num_train_epochs": 1,
|
| 157 |
+
"save_steps": 100,
|
| 158 |
+
"stateful_callbacks": {
|
| 159 |
+
"TrainerControl": {
|
| 160 |
+
"args": {
|
| 161 |
+
"should_epoch_stop": false,
|
| 162 |
+
"should_evaluate": false,
|
| 163 |
+
"should_log": false,
|
| 164 |
+
"should_save": true,
|
| 165 |
+
"should_training_stop": false
|
| 166 |
+
},
|
| 167 |
+
"attributes": {}
|
| 168 |
+
}
|
| 169 |
+
},
|
| 170 |
+
"total_flos": 2168126334566400.0,
|
| 171 |
+
"train_batch_size": 128,
|
| 172 |
+
"trial_name": null,
|
| 173 |
+
"trial_params": null
|
| 174 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bda593b8a69e45e9f48a5e12fceefb418c681431f4313d1ee1004dc70bcfa0dd
|
| 3 |
+
size 4920
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-500/config.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"activation": "silu-waleed10",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"TinyLlamaForCausalLM"
|
| 5 |
+
],
|
| 6 |
+
"attention_bias": false,
|
| 7 |
+
"attention_dropout": 0.0,
|
| 8 |
+
"bos_token_id": 1,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"eos_token_id": 2,
|
| 11 |
+
"head_dim": 32,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 128,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 256,
|
| 16 |
+
"max_position_embeddings": 512,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"mlp_type": "mlp",
|
| 19 |
+
"model_type": "tiny_llama",
|
| 20 |
+
"num_attention_heads": 4,
|
| 21 |
+
"num_hidden_layers": 21,
|
| 22 |
+
"num_key_value_heads": 4,
|
| 23 |
+
"pad_token_id": 0,
|
| 24 |
+
"pretraining_tp": 1,
|
| 25 |
+
"rms_norm_eps": 1e-06,
|
| 26 |
+
"rope_parameters": {
|
| 27 |
+
"rope_theta": 10000.0,
|
| 28 |
+
"rope_type": "default"
|
| 29 |
+
},
|
| 30 |
+
"tie_word_embeddings": true,
|
| 31 |
+
"tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
|
| 32 |
+
"transformers_version": "5.16.0.dev0",
|
| 33 |
+
"use_cache": false,
|
| 34 |
+
"vocab_size": 4096,
|
| 35 |
+
"waleed_beta": 10.0
|
| 36 |
+
}
|
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-500/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ed355b33e04badadbc8734b45e04bcc7dcb258518654b97927e25e4219617980
|
| 3 |
+
size 7959296
|