w-ahmad commited on
Commit
6651ae4
·
verified ·
1 Parent(s): b4bfbe9

Auto upload zain 2026-08-14T23:07:46.851605 (part 6)

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. zain/Activation/out/glu-waleedglu_low-21L_run/tokenizer_config.json +13 -0
  2. zain/Activation/out/glu-waleedglu_low-21L_run/training_args.bin +3 -0
  3. zain/Activation/out/glu-waleedglu_low-21L_run/training_log.jsonl +53 -0
  4. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/config.json +36 -0
  5. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/model.safetensors +3 -0
  6. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/optimizer.pt +3 -0
  7. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/rng_state.pth +3 -0
  8. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/scheduler.pt +3 -0
  9. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/tokenizer.json +0 -0
  10. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/tokenizer_config.json +13 -0
  11. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/trainer_state.json +69 -0
  12. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/training_args.bin +3 -0
  13. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/config.json +36 -0
  14. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/model.safetensors +3 -0
  15. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/optimizer.pt +3 -0
  16. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/rng_state.pth +3 -0
  17. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/scheduler.pt +3 -0
  18. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/tokenizer.json +0 -0
  19. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/tokenizer_config.json +13 -0
  20. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/trainer_state.json +392 -0
  21. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/training_args.bin +3 -0
  22. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/config.json +36 -0
  23. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/model.safetensors +3 -0
  24. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/optimizer.pt +3 -0
  25. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/rng_state.pth +3 -0
  26. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/scheduler.pt +3 -0
  27. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/tokenizer.json +0 -0
  28. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/tokenizer_config.json +13 -0
  29. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/trainer_state.json +104 -0
  30. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/training_args.bin +3 -0
  31. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/config.json +36 -0
  32. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/model.safetensors +3 -0
  33. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/optimizer.pt +3 -0
  34. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/rng_state.pth +3 -0
  35. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/scheduler.pt +3 -0
  36. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/tokenizer.json +0 -0
  37. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/tokenizer_config.json +13 -0
  38. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/trainer_state.json +139 -0
  39. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/training_args.bin +3 -0
  40. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/config.json +36 -0
  41. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/model.safetensors +3 -0
  42. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/optimizer.pt +3 -0
  43. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/rng_state.pth +3 -0
  44. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/scheduler.pt +3 -0
  45. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/tokenizer.json +0 -0
  46. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/tokenizer_config.json +13 -0
  47. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/trainer_state.json +174 -0
  48. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/training_args.bin +3 -0
  49. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-500/config.json +36 -0
  50. zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-500/model.safetensors +3 -0
zain/Activation/out/glu-waleedglu_low-21L_run/tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1024,
10
+ "pad_token": "<|endoftext|>",
11
+ "tokenizer_class": "GPT2Tokenizer",
12
+ "unk_token": "<|endoftext|>"
13
+ }
zain/Activation/out/glu-waleedglu_low-21L_run/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b710bd96acfb6390e443452d6d294cbaf327cdd790a050acb04137ff8f2a9d50
3
+ size 4920
zain/Activation/out/glu-waleedglu_low-21L_run/training_log.jsonl ADDED
@@ -0,0 +1,53 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"step": 20, "epoch": 0.010784578053383662, "timestamp": 1786747955.5507054, "loss": 28.87822570800781, "grad_norm": 4.0, "learning_rate": 0.001, "train/total_time_seconds": 8.125255100429058, "train/time_per_step_avg": 0.4062627550214529, "train/epoch_time_elapsed": 15.911321021616459, "train/estimated_remaining_minutes": 6.63562499868373}
2
+ {"step": 40, "epoch": 0.021569156106767323, "timestamp": 1786747971.392193, "loss": 24.06096496582031, "grad_norm": 0.921875, "learning_rate": 0.001, "train/total_time_seconds": 16.25177477672696, "train/time_per_step_avg": 0.40629436941817404, "train/epoch_time_elapsed": 31.752809405326843, "train/estimated_remaining_minutes": 6.500709910690785}
3
+ {"step": 60, "epoch": 0.032353734160150985, "timestamp": 1786747987.2228582, "loss": 22.317706298828124, "grad_norm": 1.4140625, "learning_rate": 0.001, "train/total_time_seconds": 24.36207727342844, "train/time_per_step_avg": 0.40603462122380735, "train/epoch_time_elapsed": 47.58347408473492, "train/estimated_remaining_minutes": 6.361209065839648}
4
+ {"step": 80, "epoch": 0.043138312213534646, "timestamp": 1786748002.850897, "loss": 20.63355407714844, "grad_norm": 2.484375, "learning_rate": 0.001, "train/total_time_seconds": 32.44940710440278, "train/time_per_step_avg": 0.40561758880503473, "train/epoch_time_elapsed": 63.2115133702755, "train/estimated_remaining_minutes": 6.219469695010533}
5
+ {"step": 100, "epoch": 0.05392289026691831, "timestamp": 1786748018.7311933, "loss": 18.89258575439453, "grad_norm": 1.8515625, "learning_rate": 0.001, "train/total_time_seconds": 40.5365272462368, "train/time_per_step_avg": 0.40536527246236803, "train/epoch_time_elapsed": 79.09180996194482, "train/estimated_remaining_minutes": 6.0804790869355205}
6
+ {"step": 120, "epoch": 0.06470746832030197, "timestamp": 1786748034.5811245, "loss": 17.514393615722657, "grad_norm": 2.140625, "learning_rate": 0.001, "train/total_time_seconds": 48.669238332659006, "train/time_per_step_avg": 0.40543983232229946, "train/epoch_time_elapsed": 94.94174122810364, "train/estimated_remaining_minutes": 5.948462462880545}
7
+ {"step": 140, "epoch": 0.07549204637368563, "timestamp": 1786748050.41628, "loss": 16.459072875976563, "grad_norm": 2.671875, "learning_rate": 0.001, "train/total_time_seconds": 56.77880089357495, "train/time_per_step_avg": 0.4052702611684799, "train/epoch_time_elapsed": 110.77689604461193, "train/estimated_remaining_minutes": 5.813067710532674}
8
+ {"step": 160, "epoch": 0.08627662442706929, "timestamp": 1786748066.2085338, "loss": 15.619828796386718, "grad_norm": 2.34375, "learning_rate": 0.001, "train/total_time_seconds": 64.8743740580976, "train/time_per_step_avg": 0.4051229678466916, "train/epoch_time_elapsed": 126.56915009394288, "train/estimated_remaining_minutes": 5.67650773008354}
9
+ {"step": 180, "epoch": 0.09706120248045295, "timestamp": 1786748082.0450277, "loss": 14.93470458984375, "grad_norm": 2.125, "learning_rate": 0.001, "train/total_time_seconds": 73.00178475677967, "train/time_per_step_avg": 0.4055237765237689, "train/epoch_time_elapsed": 142.4056449122727, "train/estimated_remaining_minutes": 5.5427281019036405}
10
+ {"step": 200, "epoch": 0.10784578053383662, "timestamp": 1786748097.7236798, "loss": 14.343766784667968, "grad_norm": 3.703125, "learning_rate": 0.001, "train/total_time_seconds": 81.08889551833272, "train/time_per_step_avg": 0.4055236827209592, "train/epoch_time_elapsed": 158.0842955261469, "train/estimated_remaining_minutes": 5.405926367888848}
11
+ {"step": 220, "epoch": 0.11863035858722028, "timestamp": 1786748113.5104978, "loss": 13.889820861816407, "grad_norm": 2.1875, "learning_rate": 0.001, "train/total_time_seconds": 89.17833111807704, "train/time_per_step_avg": 0.4050909278541803, "train/epoch_time_elapsed": 173.8711139522493, "train/estimated_remaining_minutes": 5.26962865697728}
12
+ {"step": 240, "epoch": 0.12941493664060394, "timestamp": 1786748129.3850553, "loss": 13.443467712402343, "grad_norm": 2.9375, "learning_rate": 0.001, "train/total_time_seconds": 97.44434486329556, "train/time_per_step_avg": 0.406655439697206, "train/epoch_time_elapsed": 189.74567170068622, "train/estimated_remaining_minutes": 5.142895978896154}
13
+ {"step": 260, "epoch": 0.1401995146939876, "timestamp": 1786748145.3215604, "loss": 12.998054504394531, "grad_norm": 1.6484375, "learning_rate": 0.001, "train/total_time_seconds": 105.56989968568087, "train/time_per_step_avg": 0.4069552562758327, "train/epoch_time_elapsed": 205.68217654898763, "train/estimated_remaining_minutes": 5.007802933807938}
14
+ {"step": 280, "epoch": 0.15098409274737126, "timestamp": 1786748161.172112, "loss": 12.626632690429688, "grad_norm": 2.453125, "learning_rate": 0.001, "train/total_time_seconds": 113.67667726427317, "train/time_per_step_avg": 0.406748925074935, "train/epoch_time_elapsed": 221.53272734954953, "train/estimated_remaining_minutes": 4.871857597040279}
15
+ {"step": 300, "epoch": 0.1617686708007549, "timestamp": 1786748176.9442291, "loss": 12.287302398681641, "grad_norm": 2.46875, "learning_rate": 0.001, "train/total_time_seconds": 121.78035191074014, "train/time_per_step_avg": 0.40691456392407416, "train/epoch_time_elapsed": 237.30484534427524, "train/estimated_remaining_minutes": 4.735902574306562}
16
+ {"step": 320, "epoch": 0.17255324885413859, "timestamp": 1786748192.868352, "loss": 11.972341156005859, "grad_norm": 1.765625, "learning_rate": 0.001, "train/total_time_seconds": 129.87322784215212, "train/time_per_step_avg": 0.4069489672407508, "train/epoch_time_elapsed": 253.228967346251, "train/estimated_remaining_minutes": 4.599676819409554}
17
+ {"step": 340, "epoch": 0.18333782690752223, "timestamp": 1786748208.6168966, "loss": 11.687677764892578, "grad_norm": 2.359375, "learning_rate": 0.001, "train/total_time_seconds": 137.99583918601274, "train/time_per_step_avg": 0.4055149432271719, "train/epoch_time_elapsed": 268.97751319408417, "train/estimated_remaining_minutes": 4.464571267782765}
18
+ {"step": 360, "epoch": 0.1941224049609059, "timestamp": 1786748224.4517117, "loss": 11.4271728515625, "grad_norm": 2.078125, "learning_rate": 0.001, "train/total_time_seconds": 146.12659545615315, "train/time_per_step_avg": 0.4055669577047229, "train/epoch_time_elapsed": 284.812327362597, "train/estimated_remaining_minutes": 4.329676902404538}
19
+ {"step": 380, "epoch": 0.20490698301428956, "timestamp": 1786748240.3111615, "loss": 11.163407135009766, "grad_norm": 2.0, "learning_rate": 0.001, "train/total_time_seconds": 154.2320696786046, "train/time_per_step_avg": 0.40555392414331437, "train/epoch_time_elapsed": 300.67177828401327, "train/estimated_remaining_minutes": 4.194029964944511}
20
+ {"step": 400, "epoch": 0.21569156106767323, "timestamp": 1786748256.2138653, "loss": 10.960685729980469, "grad_norm": 1.9453125, "learning_rate": 0.001, "train/total_time_seconds": 162.36926352605224, "train/time_per_step_avg": 0.405889116153121, "train/epoch_time_elapsed": 316.57448191568255, "train/estimated_remaining_minutes": 4.059231588151306}
21
+ {"step": 420, "epoch": 0.22647613912105688, "timestamp": 1786748272.037616, "loss": 10.743543243408203, "grad_norm": 1.9375, "learning_rate": 0.001, "train/total_time_seconds": 170.50094760209322, "train/time_per_step_avg": 0.406277197599411, "train/epoch_time_elapsed": 332.39823308214545, "train/estimated_remaining_minutes": 3.924228159095797}
22
+ {"step": 440, "epoch": 0.23726071717444056, "timestamp": 1786748287.8980882, "loss": 10.606502532958984, "grad_norm": 2.421875, "learning_rate": 0.001, "train/total_time_seconds": 178.61680870130658, "train/time_per_step_avg": 0.40620969515293837, "train/epoch_time_elapsed": 348.2587040960789, "train/estimated_remaining_minutes": 3.788841396694382}
23
+ {"step": 460, "epoch": 0.2480452952278242, "timestamp": 1786748303.573111, "loss": 10.407550048828124, "grad_norm": 2.375, "learning_rate": 0.001, "train/total_time_seconds": 186.72400549799204, "train/time_per_step_avg": 0.40597410041838883, "train/epoch_time_elapsed": 363.933727145195, "train/estimated_remaining_minutes": 3.6532957597433224}
24
+ {"step": 480, "epoch": 0.2588298732812079, "timestamp": 1786748319.429599, "loss": 10.243016052246094, "grad_norm": 2.109375, "learning_rate": 0.001, "train/total_time_seconds": 194.82527919858694, "train/time_per_step_avg": 0.4059320951998234, "train/epoch_time_elapsed": 379.79021544381976, "train/estimated_remaining_minutes": 3.5176786521967087}
25
+ {"step": 500, "epoch": 0.26961445133459155, "timestamp": 1786748335.175178, "loss": 10.094182586669922, "grad_norm": 1.953125, "learning_rate": 0.001, "train/total_time_seconds": 202.9594863653183, "train/time_per_step_avg": 0.4059022283926606, "train/epoch_time_elapsed": 395.5357943326235, "train/estimated_remaining_minutes": 3.3826581060886385}
26
+ {"step": 520, "epoch": 0.2803990293879752, "timestamp": 1786748351.1012967, "loss": 9.965821075439454, "grad_norm": 2.28125, "learning_rate": 0.001, "train/total_time_seconds": 211.05544440820813, "train/time_per_step_avg": 0.40554496806114915, "train/epoch_time_elapsed": 411.46191192790866, "train/estimated_remaining_minutes": 3.247006837049356}
27
+ {"step": 540, "epoch": 0.29118360744135885, "timestamp": 1786748366.8551178, "loss": 9.849958038330078, "grad_norm": 2.296875, "learning_rate": 0.001, "train/total_time_seconds": 219.1496572867036, "train/time_per_step_avg": 0.4053284858539701, "train/epoch_time_elapsed": 427.2157339602709, "train/estimated_remaining_minutes": 3.1113840232062855}
28
+ {"step": 560, "epoch": 0.3019681854947425, "timestamp": 1786748382.6123364, "loss": 9.730626678466797, "grad_norm": 2.140625, "learning_rate": 0.001, "train/total_time_seconds": 227.23986832424998, "train/time_per_step_avg": 0.4051586282625794, "train/epoch_time_elapsed": 442.9729521200061, "train/estimated_remaining_minutes": 2.9757601804366067}
29
+ {"step": 580, "epoch": 0.3127527635481262, "timestamp": 1786748398.3827922, "loss": 9.614877319335937, "grad_norm": 1.859375, "learning_rate": 0.001, "train/total_time_seconds": 235.33354591205716, "train/time_per_step_avg": 0.4050826671347022, "train/epoch_time_elapsed": 458.74340914189816, "train/estimated_remaining_minutes": 2.840232450662759}
30
+ {"step": 600, "epoch": 0.3235373416015098, "timestamp": 1786748414.3937943, "loss": 9.505727386474609, "grad_norm": 2.296875, "learning_rate": 0.001, "train/total_time_seconds": 243.46157986670732, "train/time_per_step_avg": 0.40502093501389025, "train/epoch_time_elapsed": 474.7544106952846, "train/estimated_remaining_minutes": 2.705128665185637}
31
+ {"step": 620, "epoch": 0.3343219196548935, "timestamp": 1786748430.427414, "loss": 9.414925384521485, "grad_norm": 2.046875, "learning_rate": 0.001, "train/total_time_seconds": 251.58637992665172, "train/time_per_step_avg": 0.40530935518443584, "train/epoch_time_elapsed": 490.7880291491747, "train/estimated_remaining_minutes": 2.569968397100206}
32
+ {"step": 640, "epoch": 0.34510649770827717, "timestamp": 1786748446.4542549, "loss": 9.328840637207032, "grad_norm": 1.9921875, "learning_rate": 0.001, "train/total_time_seconds": 259.7155615314841, "train/time_per_step_avg": 0.40565904244780543, "train/epoch_time_elapsed": 506.8148699365556, "train/estimated_remaining_minutes": 2.4348333893576637}
33
+ {"step": 660, "epoch": 0.35589107576166085, "timestamp": 1786748462.320152, "loss": 9.245575714111329, "grad_norm": 1.9609375, "learning_rate": 0.001, "train/total_time_seconds": 267.84740838781, "train/time_per_step_avg": 0.4060754006356001, "train/epoch_time_elapsed": 522.68076909706, "train/estimated_remaining_minutes": 2.2996999710064494}
34
+ {"step": 680, "epoch": 0.36667565381504447, "timestamp": 1786748478.2107358, "loss": 9.160669708251953, "grad_norm": 2.078125, "learning_rate": 0.001, "train/total_time_seconds": 275.9440933018923, "train/time_per_step_avg": 0.40610547389835117, "train/epoch_time_elapsed": 538.5713514313102, "train/estimated_remaining_minutes": 2.164267398446214}
35
+ {"step": 700, "epoch": 0.37746023186842814, "timestamp": 1786748493.9718537, "loss": 9.08907699584961, "grad_norm": 1.59375, "learning_rate": 0.001, "train/total_time_seconds": 284.0659934170544, "train/time_per_step_avg": 0.4060441355034709, "train/epoch_time_elapsed": 554.332470394671, "train/estimated_remaining_minutes": 2.029042810121817}
36
+ {"step": 720, "epoch": 0.3882448099218118, "timestamp": 1786748510.2042859, "loss": 8.998544311523437, "grad_norm": 1.921875, "learning_rate": 0.001, "train/total_time_seconds": 292.20687505602837, "train/time_per_step_avg": 0.4062049512937665, "train/epoch_time_elapsed": 570.5649027526379, "train/estimated_remaining_minutes": 1.8939334494372209}
37
+ {"step": 740, "epoch": 0.3990293879751955, "timestamp": 1786748526.0544066, "loss": 8.961595153808593, "grad_norm": 1.8515625, "learning_rate": 0.001, "train/total_time_seconds": 300.34223844110966, "train/time_per_step_avg": 0.4062667690962553, "train/epoch_time_elapsed": 586.4150231704116, "train/estimated_remaining_minutes": 1.7587608557362278}
38
+ {"step": 760, "epoch": 0.4098139660285791, "timestamp": 1786748541.9264393, "loss": 8.89373779296875, "grad_norm": 1.78125, "learning_rate": 0.001, "train/total_time_seconds": 308.4568819515407, "train/time_per_step_avg": 0.4060947356373072, "train/epoch_time_elapsed": 602.2870550639927, "train/estimated_remaining_minutes": 1.6234572734291617}
39
+ {"step": 780, "epoch": 0.4205985440819628, "timestamp": 1786748557.5973103, "loss": 8.858226776123047, "grad_norm": 1.90625, "learning_rate": 0.001, "train/total_time_seconds": 316.5688179396093, "train/time_per_step_avg": 0.4062472463771701, "train/epoch_time_elapsed": 617.9579257294536, "train/estimated_remaining_minutes": 1.4881440159554282}
40
+ {"step": 800, "epoch": 0.43138312213534646, "timestamp": 1786748573.3539927, "loss": 8.755474090576172, "grad_norm": 1.8359375, "learning_rate": 0.001, "train/total_time_seconds": 324.6625647544861, "train/time_per_step_avg": 0.4059657133743167, "train/epoch_time_elapsed": 633.7146084494889, "train/estimated_remaining_minutes": 1.3527606864770254}
41
+ {"step": 820, "epoch": 0.44216770018873014, "timestamp": 1786748589.0057478, "loss": 8.732923126220703, "grad_norm": 1.796875, "learning_rate": 0.001, "train/total_time_seconds": 332.75162014365196, "train/time_per_step_avg": 0.40544745087623596, "train/epoch_time_elapsed": 649.3663630895317, "train/estimated_remaining_minutes": 1.217383976135312}
42
+ {"step": 840, "epoch": 0.45295227824211376, "timestamp": 1786748604.9225793, "loss": 8.691609954833984, "grad_norm": 1.8203125, "learning_rate": 0.001, "train/total_time_seconds": 340.86094953864813, "train/time_per_step_avg": 0.40518711097538473, "train/epoch_time_elapsed": 665.2831956408918, "train/estimated_remaining_minutes": 1.0820982525036449}
43
+ {"step": 860, "epoch": 0.46373685629549743, "timestamp": 1786748620.8106012, "loss": 8.623181915283203, "grad_norm": 1.921875, "learning_rate": 0.001, "train/total_time_seconds": 348.9605983234942, "train/time_per_step_avg": 0.40503716371953485, "train/epoch_time_elapsed": 681.1712169833481, "train/estimated_remaining_minutes": 0.9467923210327363}
44
+ {"step": 880, "epoch": 0.4745214343488811, "timestamp": 1786748636.4933302, "loss": 8.585681915283203, "grad_norm": 1.84375, "learning_rate": 0.001, "train/total_time_seconds": 357.0373779311776, "train/time_per_step_avg": 0.40468559991568326, "train/epoch_time_elapsed": 696.8539464510977, "train/estimated_remaining_minutes": 0.8114485862072218}
45
+ {"step": 900, "epoch": 0.4853060124022648, "timestamp": 1786748652.084348, "loss": 8.534693145751953, "grad_norm": 1.9453125, "learning_rate": 0.001, "train/total_time_seconds": 365.11572790145874, "train/time_per_step_avg": 0.40453163146972654, "train/epoch_time_elapsed": 712.4449638240039, "train/estimated_remaining_minutes": 0.6761402368545533}
46
+ {"step": 920, "epoch": 0.4960905904556484, "timestamp": 1786748668.1569273, "loss": 8.493429565429688, "grad_norm": 1.875, "learning_rate": 0.001, "train/total_time_seconds": 373.26814346015453, "train/time_per_step_avg": 0.40516523316502573, "train/epoch_time_elapsed": 728.5175438411534, "train/estimated_remaining_minutes": 0.5409683238552964}
47
+ {"step": 940, "epoch": 0.5068751685090321, "timestamp": 1786748684.0965688, "loss": 8.474921417236327, "grad_norm": 1.640625, "learning_rate": 0.001, "train/total_time_seconds": 381.41837418451905, "train/time_per_step_avg": 0.40557424645870926, "train/epoch_time_elapsed": 744.4571851566434, "train/estimated_remaining_minutes": 0.40576422785587135}
48
+ {"step": 960, "epoch": 0.5176597465624158, "timestamp": 1786748699.8346756, "loss": 8.457220458984375, "grad_norm": 1.7890625, "learning_rate": 0.001, "train/total_time_seconds": 389.5119272880256, "train/time_per_step_avg": 0.4055132896453142, "train/epoch_time_elapsed": 760.1952919065952, "train/estimated_remaining_minutes": 0.2704943939500178}
49
+ {"step": 980, "epoch": 0.5284443246157994, "timestamp": 1786748715.6499183, "loss": 8.418341064453125, "grad_norm": 1.6953125, "learning_rate": 0.001, "train/total_time_seconds": 397.64077776297927, "train/time_per_step_avg": 0.4060339983180165, "train/epoch_time_elapsed": 776.0105344839394, "train/estimated_remaining_minutes": 0.13525196522550315}
50
+ {"step": 1000, "epoch": 0.5392289026691831, "timestamp": 1786748731.611965, "loss": 8.385391235351562, "grad_norm": 1.703125, "learning_rate": 0.001, "train/total_time_seconds": 405.79406728595495, "train/time_per_step_avg": 0.4067833938449621, "train/epoch_time_elapsed": 791.9725815802813, "train/estimated_remaining_minutes": 0.0}
51
+ {"step": 1000, "epoch": 0.5392289026691831, "timestamp": 1786748741.299589, "eval_loss": 2.096169948577881, "eval_runtime": 9.686, "eval_samples_per_second": 983.588, "eval_steps_per_second": 7.743, "train/total_time_seconds": 405.79406728595495, "train/time_per_step_avg": 0.4067833938449621, "train/epoch_time_elapsed": 801.6602005288005, "train/estimated_remaining_minutes": 0.0}
52
+ {"step": 1000, "epoch": 0.5392289026691831, "timestamp": 1786748741.3706875, "train_runtime": 802.5252, "train_samples_per_second": 637.986, "train_steps_per_second": 1.246, "total_flos": 5420315836416000.0, "train_loss": 11.859544631958007, "train/total_time_seconds": 405.79406728595495, "train/time_per_step_avg": 0.4067833938449621, "train/epoch_time_elapsed": 801.731302190572, "train/estimated_remaining_minutes": 0.0}
53
+ {"step": 1000, "epoch": 0.5392289026691831, "timestamp": 1786748751.1893418, "eval_loss": 2.096169948577881, "eval_runtime": 9.8163, "eval_samples_per_second": 970.533, "eval_steps_per_second": 7.64, "train/total_time_seconds": 405.79406728595495, "train/time_per_step_avg": 0.4067833938449621, "train/epoch_time_elapsed": 811.5499559640884, "train/estimated_remaining_minutes": 0.0}
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/config.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation": "silu-waleed10",
3
+ "architectures": [
4
+ "TinyLlamaForCausalLM"
5
+ ],
6
+ "attention_bias": false,
7
+ "attention_dropout": 0.0,
8
+ "bos_token_id": 1,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 2,
11
+ "head_dim": 32,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 128,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 256,
16
+ "max_position_embeddings": 512,
17
+ "mlp_bias": false,
18
+ "mlp_type": "mlp",
19
+ "model_type": "tiny_llama",
20
+ "num_attention_heads": 4,
21
+ "num_hidden_layers": 21,
22
+ "num_key_value_heads": 4,
23
+ "pad_token_id": 0,
24
+ "pretraining_tp": 1,
25
+ "rms_norm_eps": 1e-06,
26
+ "rope_parameters": {
27
+ "rope_theta": 10000.0,
28
+ "rope_type": "default"
29
+ },
30
+ "tie_word_embeddings": true,
31
+ "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
+ "transformers_version": "5.16.0.dev0",
33
+ "use_cache": false,
34
+ "vocab_size": 4096,
35
+ "waleed_beta": 10.0
36
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9d9b1687dfd991ff042f68918b21a9bdfd5c2562b648e54a3cdbf2d610b3a9c9
3
+ size 7959296
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f2e535ffd2007a8299d146c102ef2a448eb0bc6ac87f554927192f35c0e771b6
3
+ size 16023738
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9d9cd6a0487226e5bd30d1846894c82af483733ab4381b75bae9c0745e05d405
3
+ size 14244
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:906745ab5f61e73e8c1ba850ef3b839e2766eba889b49c843836266988895165
3
+ size 1064
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1024,
10
+ "pad_token": "<|endoftext|>",
11
+ "tokenizer_class": "GPT2Tokenizer",
12
+ "unk_token": "<|endoftext|>"
13
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/trainer_state.json ADDED
@@ -0,0 +1,69 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.05392289026691831,
6
+ "eval_steps": 60000,
7
+ "global_step": 100,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.010784578053383662,
14
+ "grad_norm": 4.09375,
15
+ "learning_rate": 0.001,
16
+ "loss": 29.103460693359374,
17
+ "step": 20
18
+ },
19
+ {
20
+ "epoch": 0.021569156106767323,
21
+ "grad_norm": 2.21875,
22
+ "learning_rate": 0.001,
23
+ "loss": 24.323658752441407,
24
+ "step": 40
25
+ },
26
+ {
27
+ "epoch": 0.032353734160150985,
28
+ "grad_norm": 1.390625,
29
+ "learning_rate": 0.001,
30
+ "loss": 22.974208068847656,
31
+ "step": 60
32
+ },
33
+ {
34
+ "epoch": 0.043138312213534646,
35
+ "grad_norm": 1.625,
36
+ "learning_rate": 0.001,
37
+ "loss": 21.970303344726563,
38
+ "step": 80
39
+ },
40
+ {
41
+ "epoch": 0.05392289026691831,
42
+ "grad_norm": 3.625,
43
+ "learning_rate": 0.001,
44
+ "loss": 20.951182556152343,
45
+ "step": 100
46
+ }
47
+ ],
48
+ "logging_steps": 20,
49
+ "max_steps": 1000,
50
+ "num_input_tokens_seen": 0,
51
+ "num_train_epochs": 1,
52
+ "save_steps": 100,
53
+ "stateful_callbacks": {
54
+ "TrainerControl": {
55
+ "args": {
56
+ "should_epoch_stop": false,
57
+ "should_evaluate": false,
58
+ "should_log": false,
59
+ "should_save": true,
60
+ "should_training_stop": false
61
+ },
62
+ "attributes": {}
63
+ }
64
+ },
65
+ "total_flos": 542031583641600.0,
66
+ "train_batch_size": 128,
67
+ "trial_name": null,
68
+ "trial_params": null
69
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-100/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bda593b8a69e45e9f48a5e12fceefb418c681431f4313d1ee1004dc70bcfa0dd
3
+ size 4920
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/config.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation": "silu-waleed10",
3
+ "architectures": [
4
+ "TinyLlamaForCausalLM"
5
+ ],
6
+ "attention_bias": false,
7
+ "attention_dropout": 0.0,
8
+ "bos_token_id": 1,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 2,
11
+ "head_dim": 32,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 128,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 256,
16
+ "max_position_embeddings": 512,
17
+ "mlp_bias": false,
18
+ "mlp_type": "mlp",
19
+ "model_type": "tiny_llama",
20
+ "num_attention_heads": 4,
21
+ "num_hidden_layers": 21,
22
+ "num_key_value_heads": 4,
23
+ "pad_token_id": 0,
24
+ "pretraining_tp": 1,
25
+ "rms_norm_eps": 1e-06,
26
+ "rope_parameters": {
27
+ "rope_theta": 10000.0,
28
+ "rope_type": "default"
29
+ },
30
+ "tie_word_embeddings": true,
31
+ "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
+ "transformers_version": "5.16.0.dev0",
33
+ "use_cache": false,
34
+ "vocab_size": 4096,
35
+ "waleed_beta": 10.0
36
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0ba794fb6fcee5b52be8662d141115e7a75521492fb734a5fb441096c5899460
3
+ size 7959296
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4a30891a43b0159714e9af5b848f983099e515a8c934604658fe059bf1e5e70b
3
+ size 16023738
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2b66e3cc7c452b707ddac5caf0aa17618afb9bc1a0333600a22c4afb353f3165
3
+ size 14244
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0fc3ca8af69cc9d003a05a139e7997960ab393fe31ae1d898b29038573704b4e
3
+ size 1064
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1024,
10
+ "pad_token": "<|endoftext|>",
11
+ "tokenizer_class": "GPT2Tokenizer",
12
+ "unk_token": "<|endoftext|>"
13
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/trainer_state.json ADDED
@@ -0,0 +1,392 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.5392289026691831,
6
+ "eval_steps": 60000,
7
+ "global_step": 1000,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.010784578053383662,
14
+ "grad_norm": 4.09375,
15
+ "learning_rate": 0.001,
16
+ "loss": 29.103460693359374,
17
+ "step": 20
18
+ },
19
+ {
20
+ "epoch": 0.021569156106767323,
21
+ "grad_norm": 2.21875,
22
+ "learning_rate": 0.001,
23
+ "loss": 24.323658752441407,
24
+ "step": 40
25
+ },
26
+ {
27
+ "epoch": 0.032353734160150985,
28
+ "grad_norm": 1.390625,
29
+ "learning_rate": 0.001,
30
+ "loss": 22.974208068847656,
31
+ "step": 60
32
+ },
33
+ {
34
+ "epoch": 0.043138312213534646,
35
+ "grad_norm": 1.625,
36
+ "learning_rate": 0.001,
37
+ "loss": 21.970303344726563,
38
+ "step": 80
39
+ },
40
+ {
41
+ "epoch": 0.05392289026691831,
42
+ "grad_norm": 3.625,
43
+ "learning_rate": 0.001,
44
+ "loss": 20.951182556152343,
45
+ "step": 100
46
+ },
47
+ {
48
+ "epoch": 0.06470746832030197,
49
+ "grad_norm": 3.375,
50
+ "learning_rate": 0.001,
51
+ "loss": 20.1780517578125,
52
+ "step": 120
53
+ },
54
+ {
55
+ "epoch": 0.07549204637368563,
56
+ "grad_norm": 3.796875,
57
+ "learning_rate": 0.001,
58
+ "loss": 19.392193603515626,
59
+ "step": 140
60
+ },
61
+ {
62
+ "epoch": 0.08627662442706929,
63
+ "grad_norm": 3.21875,
64
+ "learning_rate": 0.001,
65
+ "loss": 18.65845489501953,
66
+ "step": 160
67
+ },
68
+ {
69
+ "epoch": 0.09706120248045295,
70
+ "grad_norm": 2.609375,
71
+ "learning_rate": 0.001,
72
+ "loss": 17.959799194335936,
73
+ "step": 180
74
+ },
75
+ {
76
+ "epoch": 0.10784578053383662,
77
+ "grad_norm": 2.359375,
78
+ "learning_rate": 0.001,
79
+ "loss": 17.317408752441406,
80
+ "step": 200
81
+ },
82
+ {
83
+ "epoch": 0.11863035858722028,
84
+ "grad_norm": 3.484375,
85
+ "learning_rate": 0.001,
86
+ "loss": 16.665341186523438,
87
+ "step": 220
88
+ },
89
+ {
90
+ "epoch": 0.12941493664060394,
91
+ "grad_norm": 4.0625,
92
+ "learning_rate": 0.001,
93
+ "loss": 16.151258850097655,
94
+ "step": 240
95
+ },
96
+ {
97
+ "epoch": 0.1401995146939876,
98
+ "grad_norm": 2.34375,
99
+ "learning_rate": 0.001,
100
+ "loss": 15.6710205078125,
101
+ "step": 260
102
+ },
103
+ {
104
+ "epoch": 0.15098409274737126,
105
+ "grad_norm": 2.984375,
106
+ "learning_rate": 0.001,
107
+ "loss": 15.264056396484374,
108
+ "step": 280
109
+ },
110
+ {
111
+ "epoch": 0.1617686708007549,
112
+ "grad_norm": 1.8125,
113
+ "learning_rate": 0.001,
114
+ "loss": 14.908853149414062,
115
+ "step": 300
116
+ },
117
+ {
118
+ "epoch": 0.17255324885413859,
119
+ "grad_norm": 2.390625,
120
+ "learning_rate": 0.001,
121
+ "loss": 14.611824035644531,
122
+ "step": 320
123
+ },
124
+ {
125
+ "epoch": 0.18333782690752223,
126
+ "grad_norm": 1.6484375,
127
+ "learning_rate": 0.001,
128
+ "loss": 14.270545959472656,
129
+ "step": 340
130
+ },
131
+ {
132
+ "epoch": 0.1941224049609059,
133
+ "grad_norm": 2.03125,
134
+ "learning_rate": 0.001,
135
+ "loss": 13.935658264160157,
136
+ "step": 360
137
+ },
138
+ {
139
+ "epoch": 0.20490698301428956,
140
+ "grad_norm": 2.75,
141
+ "learning_rate": 0.001,
142
+ "loss": 13.626214599609375,
143
+ "step": 380
144
+ },
145
+ {
146
+ "epoch": 0.21569156106767323,
147
+ "grad_norm": 2.6875,
148
+ "learning_rate": 0.001,
149
+ "loss": 13.423133850097656,
150
+ "step": 400
151
+ },
152
+ {
153
+ "epoch": 0.22647613912105688,
154
+ "grad_norm": 2.640625,
155
+ "learning_rate": 0.001,
156
+ "loss": 13.1606201171875,
157
+ "step": 420
158
+ },
159
+ {
160
+ "epoch": 0.23726071717444056,
161
+ "grad_norm": 2.03125,
162
+ "learning_rate": 0.001,
163
+ "loss": 12.97046356201172,
164
+ "step": 440
165
+ },
166
+ {
167
+ "epoch": 0.2480452952278242,
168
+ "grad_norm": 2.375,
169
+ "learning_rate": 0.001,
170
+ "loss": 12.767843627929688,
171
+ "step": 460
172
+ },
173
+ {
174
+ "epoch": 0.2588298732812079,
175
+ "grad_norm": 1.9609375,
176
+ "learning_rate": 0.001,
177
+ "loss": 12.555582427978516,
178
+ "step": 480
179
+ },
180
+ {
181
+ "epoch": 0.26961445133459155,
182
+ "grad_norm": 1.6640625,
183
+ "learning_rate": 0.001,
184
+ "loss": 12.363282012939454,
185
+ "step": 500
186
+ },
187
+ {
188
+ "epoch": 0.2803990293879752,
189
+ "grad_norm": 2.40625,
190
+ "learning_rate": 0.001,
191
+ "loss": 12.229927825927735,
192
+ "step": 520
193
+ },
194
+ {
195
+ "epoch": 0.29118360744135885,
196
+ "grad_norm": 2.1875,
197
+ "learning_rate": 0.001,
198
+ "loss": 12.06595230102539,
199
+ "step": 540
200
+ },
201
+ {
202
+ "epoch": 0.3019681854947425,
203
+ "grad_norm": 2.25,
204
+ "learning_rate": 0.001,
205
+ "loss": 11.916503143310546,
206
+ "step": 560
207
+ },
208
+ {
209
+ "epoch": 0.3127527635481262,
210
+ "grad_norm": 2.0625,
211
+ "learning_rate": 0.001,
212
+ "loss": 11.778754425048827,
213
+ "step": 580
214
+ },
215
+ {
216
+ "epoch": 0.3235373416015098,
217
+ "grad_norm": 2.5625,
218
+ "learning_rate": 0.001,
219
+ "loss": 11.674578857421874,
220
+ "step": 600
221
+ },
222
+ {
223
+ "epoch": 0.3343219196548935,
224
+ "grad_norm": 1.9921875,
225
+ "learning_rate": 0.001,
226
+ "loss": 11.54565658569336,
227
+ "step": 620
228
+ },
229
+ {
230
+ "epoch": 0.34510649770827717,
231
+ "grad_norm": 2.0,
232
+ "learning_rate": 0.001,
233
+ "loss": 11.434559631347657,
234
+ "step": 640
235
+ },
236
+ {
237
+ "epoch": 0.35589107576166085,
238
+ "grad_norm": 2.203125,
239
+ "learning_rate": 0.001,
240
+ "loss": 11.319635772705078,
241
+ "step": 660
242
+ },
243
+ {
244
+ "epoch": 0.36667565381504447,
245
+ "grad_norm": 2.15625,
246
+ "learning_rate": 0.001,
247
+ "loss": 11.202098846435547,
248
+ "step": 680
249
+ },
250
+ {
251
+ "epoch": 0.37746023186842814,
252
+ "grad_norm": 2.421875,
253
+ "learning_rate": 0.001,
254
+ "loss": 11.103318023681641,
255
+ "step": 700
256
+ },
257
+ {
258
+ "epoch": 0.3882448099218118,
259
+ "grad_norm": 2.03125,
260
+ "learning_rate": 0.001,
261
+ "loss": 10.984959411621094,
262
+ "step": 720
263
+ },
264
+ {
265
+ "epoch": 0.3990293879751955,
266
+ "grad_norm": 2.78125,
267
+ "learning_rate": 0.001,
268
+ "loss": 10.932718658447266,
269
+ "step": 740
270
+ },
271
+ {
272
+ "epoch": 0.4098139660285791,
273
+ "grad_norm": 2.265625,
274
+ "learning_rate": 0.001,
275
+ "loss": 10.841696166992188,
276
+ "step": 760
277
+ },
278
+ {
279
+ "epoch": 0.4205985440819628,
280
+ "grad_norm": 2.53125,
281
+ "learning_rate": 0.001,
282
+ "loss": 10.763890838623047,
283
+ "step": 780
284
+ },
285
+ {
286
+ "epoch": 0.43138312213534646,
287
+ "grad_norm": 2.375,
288
+ "learning_rate": 0.001,
289
+ "loss": 10.633779907226563,
290
+ "step": 800
291
+ },
292
+ {
293
+ "epoch": 0.44216770018873014,
294
+ "grad_norm": 2.46875,
295
+ "learning_rate": 0.001,
296
+ "loss": 10.587508392333984,
297
+ "step": 820
298
+ },
299
+ {
300
+ "epoch": 0.45295227824211376,
301
+ "grad_norm": 2.53125,
302
+ "learning_rate": 0.001,
303
+ "loss": 10.518219757080079,
304
+ "step": 840
305
+ },
306
+ {
307
+ "epoch": 0.46373685629549743,
308
+ "grad_norm": 2.484375,
309
+ "learning_rate": 0.001,
310
+ "loss": 10.428038787841796,
311
+ "step": 860
312
+ },
313
+ {
314
+ "epoch": 0.4745214343488811,
315
+ "grad_norm": 2.234375,
316
+ "learning_rate": 0.001,
317
+ "loss": 10.359873199462891,
318
+ "step": 880
319
+ },
320
+ {
321
+ "epoch": 0.4853060124022648,
322
+ "grad_norm": 2.640625,
323
+ "learning_rate": 0.001,
324
+ "loss": 10.303240966796874,
325
+ "step": 900
326
+ },
327
+ {
328
+ "epoch": 0.4960905904556484,
329
+ "grad_norm": 2.703125,
330
+ "learning_rate": 0.001,
331
+ "loss": 10.233261108398438,
332
+ "step": 920
333
+ },
334
+ {
335
+ "epoch": 0.5068751685090321,
336
+ "grad_norm": 2.984375,
337
+ "learning_rate": 0.001,
338
+ "loss": 10.198040008544922,
339
+ "step": 940
340
+ },
341
+ {
342
+ "epoch": 0.5176597465624158,
343
+ "grad_norm": 2.234375,
344
+ "learning_rate": 0.001,
345
+ "loss": 10.153294372558594,
346
+ "step": 960
347
+ },
348
+ {
349
+ "epoch": 0.5284443246157994,
350
+ "grad_norm": 2.46875,
351
+ "learning_rate": 0.001,
352
+ "loss": 10.112586212158202,
353
+ "step": 980
354
+ },
355
+ {
356
+ "epoch": 0.5392289026691831,
357
+ "grad_norm": 2.609375,
358
+ "learning_rate": 0.001,
359
+ "loss": 10.053606414794922,
360
+ "step": 1000
361
+ },
362
+ {
363
+ "epoch": 0.5392289026691831,
364
+ "eval_loss": 2.5052456855773926,
365
+ "eval_runtime": 9.4135,
366
+ "eval_samples_per_second": 1012.06,
367
+ "eval_steps_per_second": 7.967,
368
+ "step": 1000
369
+ }
370
+ ],
371
+ "logging_steps": 20,
372
+ "max_steps": 1000,
373
+ "num_input_tokens_seen": 0,
374
+ "num_train_epochs": 1,
375
+ "save_steps": 100,
376
+ "stateful_callbacks": {
377
+ "TrainerControl": {
378
+ "args": {
379
+ "should_epoch_stop": false,
380
+ "should_evaluate": false,
381
+ "should_log": false,
382
+ "should_save": true,
383
+ "should_training_stop": true
384
+ },
385
+ "attributes": {}
386
+ }
387
+ },
388
+ "total_flos": 5420315836416000.0,
389
+ "train_batch_size": 128,
390
+ "trial_name": null,
391
+ "trial_params": null
392
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-1000/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bda593b8a69e45e9f48a5e12fceefb418c681431f4313d1ee1004dc70bcfa0dd
3
+ size 4920
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/config.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation": "silu-waleed10",
3
+ "architectures": [
4
+ "TinyLlamaForCausalLM"
5
+ ],
6
+ "attention_bias": false,
7
+ "attention_dropout": 0.0,
8
+ "bos_token_id": 1,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 2,
11
+ "head_dim": 32,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 128,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 256,
16
+ "max_position_embeddings": 512,
17
+ "mlp_bias": false,
18
+ "mlp_type": "mlp",
19
+ "model_type": "tiny_llama",
20
+ "num_attention_heads": 4,
21
+ "num_hidden_layers": 21,
22
+ "num_key_value_heads": 4,
23
+ "pad_token_id": 0,
24
+ "pretraining_tp": 1,
25
+ "rms_norm_eps": 1e-06,
26
+ "rope_parameters": {
27
+ "rope_theta": 10000.0,
28
+ "rope_type": "default"
29
+ },
30
+ "tie_word_embeddings": true,
31
+ "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
+ "transformers_version": "5.16.0.dev0",
33
+ "use_cache": false,
34
+ "vocab_size": 4096,
35
+ "waleed_beta": 10.0
36
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e47052a6a249ebf53373e960f46db44c16badf55d0b1f5641d7b6fe7bcd56add
3
+ size 7959296
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a689448d286470826d3e90b92a69afbf00fcf27b42685e171aceb475c52d12bc
3
+ size 16023738
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9d9cd6a0487226e5bd30d1846894c82af483733ab4381b75bae9c0745e05d405
3
+ size 14244
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2d3b5a970fc92af42664e84ba686059eaaacfaa668691c0cf6d2afddda677461
3
+ size 1064
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1024,
10
+ "pad_token": "<|endoftext|>",
11
+ "tokenizer_class": "GPT2Tokenizer",
12
+ "unk_token": "<|endoftext|>"
13
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/trainer_state.json ADDED
@@ -0,0 +1,104 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.10784578053383662,
6
+ "eval_steps": 60000,
7
+ "global_step": 200,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.010784578053383662,
14
+ "grad_norm": 4.09375,
15
+ "learning_rate": 0.001,
16
+ "loss": 29.103460693359374,
17
+ "step": 20
18
+ },
19
+ {
20
+ "epoch": 0.021569156106767323,
21
+ "grad_norm": 2.21875,
22
+ "learning_rate": 0.001,
23
+ "loss": 24.323658752441407,
24
+ "step": 40
25
+ },
26
+ {
27
+ "epoch": 0.032353734160150985,
28
+ "grad_norm": 1.390625,
29
+ "learning_rate": 0.001,
30
+ "loss": 22.974208068847656,
31
+ "step": 60
32
+ },
33
+ {
34
+ "epoch": 0.043138312213534646,
35
+ "grad_norm": 1.625,
36
+ "learning_rate": 0.001,
37
+ "loss": 21.970303344726563,
38
+ "step": 80
39
+ },
40
+ {
41
+ "epoch": 0.05392289026691831,
42
+ "grad_norm": 3.625,
43
+ "learning_rate": 0.001,
44
+ "loss": 20.951182556152343,
45
+ "step": 100
46
+ },
47
+ {
48
+ "epoch": 0.06470746832030197,
49
+ "grad_norm": 3.375,
50
+ "learning_rate": 0.001,
51
+ "loss": 20.1780517578125,
52
+ "step": 120
53
+ },
54
+ {
55
+ "epoch": 0.07549204637368563,
56
+ "grad_norm": 3.796875,
57
+ "learning_rate": 0.001,
58
+ "loss": 19.392193603515626,
59
+ "step": 140
60
+ },
61
+ {
62
+ "epoch": 0.08627662442706929,
63
+ "grad_norm": 3.21875,
64
+ "learning_rate": 0.001,
65
+ "loss": 18.65845489501953,
66
+ "step": 160
67
+ },
68
+ {
69
+ "epoch": 0.09706120248045295,
70
+ "grad_norm": 2.609375,
71
+ "learning_rate": 0.001,
72
+ "loss": 17.959799194335936,
73
+ "step": 180
74
+ },
75
+ {
76
+ "epoch": 0.10784578053383662,
77
+ "grad_norm": 2.359375,
78
+ "learning_rate": 0.001,
79
+ "loss": 17.317408752441406,
80
+ "step": 200
81
+ }
82
+ ],
83
+ "logging_steps": 20,
84
+ "max_steps": 1000,
85
+ "num_input_tokens_seen": 0,
86
+ "num_train_epochs": 1,
87
+ "save_steps": 100,
88
+ "stateful_callbacks": {
89
+ "TrainerControl": {
90
+ "args": {
91
+ "should_epoch_stop": false,
92
+ "should_evaluate": false,
93
+ "should_log": false,
94
+ "should_save": true,
95
+ "should_training_stop": false
96
+ },
97
+ "attributes": {}
98
+ }
99
+ },
100
+ "total_flos": 1084063167283200.0,
101
+ "train_batch_size": 128,
102
+ "trial_name": null,
103
+ "trial_params": null
104
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-200/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bda593b8a69e45e9f48a5e12fceefb418c681431f4313d1ee1004dc70bcfa0dd
3
+ size 4920
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/config.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation": "silu-waleed10",
3
+ "architectures": [
4
+ "TinyLlamaForCausalLM"
5
+ ],
6
+ "attention_bias": false,
7
+ "attention_dropout": 0.0,
8
+ "bos_token_id": 1,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 2,
11
+ "head_dim": 32,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 128,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 256,
16
+ "max_position_embeddings": 512,
17
+ "mlp_bias": false,
18
+ "mlp_type": "mlp",
19
+ "model_type": "tiny_llama",
20
+ "num_attention_heads": 4,
21
+ "num_hidden_layers": 21,
22
+ "num_key_value_heads": 4,
23
+ "pad_token_id": 0,
24
+ "pretraining_tp": 1,
25
+ "rms_norm_eps": 1e-06,
26
+ "rope_parameters": {
27
+ "rope_theta": 10000.0,
28
+ "rope_type": "default"
29
+ },
30
+ "tie_word_embeddings": true,
31
+ "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
+ "transformers_version": "5.16.0.dev0",
33
+ "use_cache": false,
34
+ "vocab_size": 4096,
35
+ "waleed_beta": 10.0
36
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f515f27d365a121e29deb1fb378e30dd334e121d4cb62d906d24656b71f2113e
3
+ size 7959296
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:326ff18127cbac66b90e9041ed245dfd03d7ad9b8654dc096a075bfe445183a6
3
+ size 16023738
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9d9cd6a0487226e5bd30d1846894c82af483733ab4381b75bae9c0745e05d405
3
+ size 14244
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:76cd21f32221b90fb8288fd2428595ec8b5039fe2425086d67cacb5c8d814c9e
3
+ size 1064
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1024,
10
+ "pad_token": "<|endoftext|>",
11
+ "tokenizer_class": "GPT2Tokenizer",
12
+ "unk_token": "<|endoftext|>"
13
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/trainer_state.json ADDED
@@ -0,0 +1,139 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.1617686708007549,
6
+ "eval_steps": 60000,
7
+ "global_step": 300,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.010784578053383662,
14
+ "grad_norm": 4.09375,
15
+ "learning_rate": 0.001,
16
+ "loss": 29.103460693359374,
17
+ "step": 20
18
+ },
19
+ {
20
+ "epoch": 0.021569156106767323,
21
+ "grad_norm": 2.21875,
22
+ "learning_rate": 0.001,
23
+ "loss": 24.323658752441407,
24
+ "step": 40
25
+ },
26
+ {
27
+ "epoch": 0.032353734160150985,
28
+ "grad_norm": 1.390625,
29
+ "learning_rate": 0.001,
30
+ "loss": 22.974208068847656,
31
+ "step": 60
32
+ },
33
+ {
34
+ "epoch": 0.043138312213534646,
35
+ "grad_norm": 1.625,
36
+ "learning_rate": 0.001,
37
+ "loss": 21.970303344726563,
38
+ "step": 80
39
+ },
40
+ {
41
+ "epoch": 0.05392289026691831,
42
+ "grad_norm": 3.625,
43
+ "learning_rate": 0.001,
44
+ "loss": 20.951182556152343,
45
+ "step": 100
46
+ },
47
+ {
48
+ "epoch": 0.06470746832030197,
49
+ "grad_norm": 3.375,
50
+ "learning_rate": 0.001,
51
+ "loss": 20.1780517578125,
52
+ "step": 120
53
+ },
54
+ {
55
+ "epoch": 0.07549204637368563,
56
+ "grad_norm": 3.796875,
57
+ "learning_rate": 0.001,
58
+ "loss": 19.392193603515626,
59
+ "step": 140
60
+ },
61
+ {
62
+ "epoch": 0.08627662442706929,
63
+ "grad_norm": 3.21875,
64
+ "learning_rate": 0.001,
65
+ "loss": 18.65845489501953,
66
+ "step": 160
67
+ },
68
+ {
69
+ "epoch": 0.09706120248045295,
70
+ "grad_norm": 2.609375,
71
+ "learning_rate": 0.001,
72
+ "loss": 17.959799194335936,
73
+ "step": 180
74
+ },
75
+ {
76
+ "epoch": 0.10784578053383662,
77
+ "grad_norm": 2.359375,
78
+ "learning_rate": 0.001,
79
+ "loss": 17.317408752441406,
80
+ "step": 200
81
+ },
82
+ {
83
+ "epoch": 0.11863035858722028,
84
+ "grad_norm": 3.484375,
85
+ "learning_rate": 0.001,
86
+ "loss": 16.665341186523438,
87
+ "step": 220
88
+ },
89
+ {
90
+ "epoch": 0.12941493664060394,
91
+ "grad_norm": 4.0625,
92
+ "learning_rate": 0.001,
93
+ "loss": 16.151258850097655,
94
+ "step": 240
95
+ },
96
+ {
97
+ "epoch": 0.1401995146939876,
98
+ "grad_norm": 2.34375,
99
+ "learning_rate": 0.001,
100
+ "loss": 15.6710205078125,
101
+ "step": 260
102
+ },
103
+ {
104
+ "epoch": 0.15098409274737126,
105
+ "grad_norm": 2.984375,
106
+ "learning_rate": 0.001,
107
+ "loss": 15.264056396484374,
108
+ "step": 280
109
+ },
110
+ {
111
+ "epoch": 0.1617686708007549,
112
+ "grad_norm": 1.8125,
113
+ "learning_rate": 0.001,
114
+ "loss": 14.908853149414062,
115
+ "step": 300
116
+ }
117
+ ],
118
+ "logging_steps": 20,
119
+ "max_steps": 1000,
120
+ "num_input_tokens_seen": 0,
121
+ "num_train_epochs": 1,
122
+ "save_steps": 100,
123
+ "stateful_callbacks": {
124
+ "TrainerControl": {
125
+ "args": {
126
+ "should_epoch_stop": false,
127
+ "should_evaluate": false,
128
+ "should_log": false,
129
+ "should_save": true,
130
+ "should_training_stop": false
131
+ },
132
+ "attributes": {}
133
+ }
134
+ },
135
+ "total_flos": 1626094750924800.0,
136
+ "train_batch_size": 128,
137
+ "trial_name": null,
138
+ "trial_params": null
139
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-300/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bda593b8a69e45e9f48a5e12fceefb418c681431f4313d1ee1004dc70bcfa0dd
3
+ size 4920
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/config.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation": "silu-waleed10",
3
+ "architectures": [
4
+ "TinyLlamaForCausalLM"
5
+ ],
6
+ "attention_bias": false,
7
+ "attention_dropout": 0.0,
8
+ "bos_token_id": 1,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 2,
11
+ "head_dim": 32,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 128,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 256,
16
+ "max_position_embeddings": 512,
17
+ "mlp_bias": false,
18
+ "mlp_type": "mlp",
19
+ "model_type": "tiny_llama",
20
+ "num_attention_heads": 4,
21
+ "num_hidden_layers": 21,
22
+ "num_key_value_heads": 4,
23
+ "pad_token_id": 0,
24
+ "pretraining_tp": 1,
25
+ "rms_norm_eps": 1e-06,
26
+ "rope_parameters": {
27
+ "rope_theta": 10000.0,
28
+ "rope_type": "default"
29
+ },
30
+ "tie_word_embeddings": true,
31
+ "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
+ "transformers_version": "5.16.0.dev0",
33
+ "use_cache": false,
34
+ "vocab_size": 4096,
35
+ "waleed_beta": 10.0
36
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8a38c2c8a915b9b6f40c28c42f072a644b4cf5db90c25784335b26df0418b09b
3
+ size 7959296
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4855eab9d35bda1625a50a9b1b82dd1490b389965177b34af7bab3d07f5e0396
3
+ size 16023738
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9d9cd6a0487226e5bd30d1846894c82af483733ab4381b75bae9c0745e05d405
3
+ size 14244
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:51769a11c87b2b665cbe64c58b934afdb1fa1998bffb4addcc2f852b172d6681
3
+ size 1064
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1024,
10
+ "pad_token": "<|endoftext|>",
11
+ "tokenizer_class": "GPT2Tokenizer",
12
+ "unk_token": "<|endoftext|>"
13
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/trainer_state.json ADDED
@@ -0,0 +1,174 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.21569156106767323,
6
+ "eval_steps": 60000,
7
+ "global_step": 400,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.010784578053383662,
14
+ "grad_norm": 4.09375,
15
+ "learning_rate": 0.001,
16
+ "loss": 29.103460693359374,
17
+ "step": 20
18
+ },
19
+ {
20
+ "epoch": 0.021569156106767323,
21
+ "grad_norm": 2.21875,
22
+ "learning_rate": 0.001,
23
+ "loss": 24.323658752441407,
24
+ "step": 40
25
+ },
26
+ {
27
+ "epoch": 0.032353734160150985,
28
+ "grad_norm": 1.390625,
29
+ "learning_rate": 0.001,
30
+ "loss": 22.974208068847656,
31
+ "step": 60
32
+ },
33
+ {
34
+ "epoch": 0.043138312213534646,
35
+ "grad_norm": 1.625,
36
+ "learning_rate": 0.001,
37
+ "loss": 21.970303344726563,
38
+ "step": 80
39
+ },
40
+ {
41
+ "epoch": 0.05392289026691831,
42
+ "grad_norm": 3.625,
43
+ "learning_rate": 0.001,
44
+ "loss": 20.951182556152343,
45
+ "step": 100
46
+ },
47
+ {
48
+ "epoch": 0.06470746832030197,
49
+ "grad_norm": 3.375,
50
+ "learning_rate": 0.001,
51
+ "loss": 20.1780517578125,
52
+ "step": 120
53
+ },
54
+ {
55
+ "epoch": 0.07549204637368563,
56
+ "grad_norm": 3.796875,
57
+ "learning_rate": 0.001,
58
+ "loss": 19.392193603515626,
59
+ "step": 140
60
+ },
61
+ {
62
+ "epoch": 0.08627662442706929,
63
+ "grad_norm": 3.21875,
64
+ "learning_rate": 0.001,
65
+ "loss": 18.65845489501953,
66
+ "step": 160
67
+ },
68
+ {
69
+ "epoch": 0.09706120248045295,
70
+ "grad_norm": 2.609375,
71
+ "learning_rate": 0.001,
72
+ "loss": 17.959799194335936,
73
+ "step": 180
74
+ },
75
+ {
76
+ "epoch": 0.10784578053383662,
77
+ "grad_norm": 2.359375,
78
+ "learning_rate": 0.001,
79
+ "loss": 17.317408752441406,
80
+ "step": 200
81
+ },
82
+ {
83
+ "epoch": 0.11863035858722028,
84
+ "grad_norm": 3.484375,
85
+ "learning_rate": 0.001,
86
+ "loss": 16.665341186523438,
87
+ "step": 220
88
+ },
89
+ {
90
+ "epoch": 0.12941493664060394,
91
+ "grad_norm": 4.0625,
92
+ "learning_rate": 0.001,
93
+ "loss": 16.151258850097655,
94
+ "step": 240
95
+ },
96
+ {
97
+ "epoch": 0.1401995146939876,
98
+ "grad_norm": 2.34375,
99
+ "learning_rate": 0.001,
100
+ "loss": 15.6710205078125,
101
+ "step": 260
102
+ },
103
+ {
104
+ "epoch": 0.15098409274737126,
105
+ "grad_norm": 2.984375,
106
+ "learning_rate": 0.001,
107
+ "loss": 15.264056396484374,
108
+ "step": 280
109
+ },
110
+ {
111
+ "epoch": 0.1617686708007549,
112
+ "grad_norm": 1.8125,
113
+ "learning_rate": 0.001,
114
+ "loss": 14.908853149414062,
115
+ "step": 300
116
+ },
117
+ {
118
+ "epoch": 0.17255324885413859,
119
+ "grad_norm": 2.390625,
120
+ "learning_rate": 0.001,
121
+ "loss": 14.611824035644531,
122
+ "step": 320
123
+ },
124
+ {
125
+ "epoch": 0.18333782690752223,
126
+ "grad_norm": 1.6484375,
127
+ "learning_rate": 0.001,
128
+ "loss": 14.270545959472656,
129
+ "step": 340
130
+ },
131
+ {
132
+ "epoch": 0.1941224049609059,
133
+ "grad_norm": 2.03125,
134
+ "learning_rate": 0.001,
135
+ "loss": 13.935658264160157,
136
+ "step": 360
137
+ },
138
+ {
139
+ "epoch": 0.20490698301428956,
140
+ "grad_norm": 2.75,
141
+ "learning_rate": 0.001,
142
+ "loss": 13.626214599609375,
143
+ "step": 380
144
+ },
145
+ {
146
+ "epoch": 0.21569156106767323,
147
+ "grad_norm": 2.6875,
148
+ "learning_rate": 0.001,
149
+ "loss": 13.423133850097656,
150
+ "step": 400
151
+ }
152
+ ],
153
+ "logging_steps": 20,
154
+ "max_steps": 1000,
155
+ "num_input_tokens_seen": 0,
156
+ "num_train_epochs": 1,
157
+ "save_steps": 100,
158
+ "stateful_callbacks": {
159
+ "TrainerControl": {
160
+ "args": {
161
+ "should_epoch_stop": false,
162
+ "should_evaluate": false,
163
+ "should_log": false,
164
+ "should_save": true,
165
+ "should_training_stop": false
166
+ },
167
+ "attributes": {}
168
+ }
169
+ },
170
+ "total_flos": 2168126334566400.0,
171
+ "train_batch_size": 128,
172
+ "trial_name": null,
173
+ "trial_params": null
174
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-400/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bda593b8a69e45e9f48a5e12fceefb418c681431f4313d1ee1004dc70bcfa0dd
3
+ size 4920
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-500/config.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation": "silu-waleed10",
3
+ "architectures": [
4
+ "TinyLlamaForCausalLM"
5
+ ],
6
+ "attention_bias": false,
7
+ "attention_dropout": 0.0,
8
+ "bos_token_id": 1,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 2,
11
+ "head_dim": 32,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 128,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 256,
16
+ "max_position_embeddings": 512,
17
+ "mlp_bias": false,
18
+ "mlp_type": "mlp",
19
+ "model_type": "tiny_llama",
20
+ "num_attention_heads": 4,
21
+ "num_hidden_layers": 21,
22
+ "num_key_value_heads": 4,
23
+ "pad_token_id": 0,
24
+ "pretraining_tp": 1,
25
+ "rms_norm_eps": 1e-06,
26
+ "rope_parameters": {
27
+ "rope_theta": 10000.0,
28
+ "rope_type": "default"
29
+ },
30
+ "tie_word_embeddings": true,
31
+ "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
+ "transformers_version": "5.16.0.dev0",
33
+ "use_cache": false,
34
+ "vocab_size": 4096,
35
+ "waleed_beta": 10.0
36
+ }
zain/Activation/out/mlp-silu-waleed10-21L_run/checkpoint-500/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ed355b33e04badadbc8734b45e04bcc7dcb258518654b97927e25e4219617980
3
+ size 7959296