w-ahmad commited on
Commit
9b4c79a
·
verified ·
1 Parent(s): 841bb5d

Auto upload zain 2026-08-12T23:51:54.973102 (part 2)

Browse files
zain/Activation/out/sweep_summary.json ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "variant": "glu-silu-waleed10-100L",
4
+ "eval_loss": 2.122779130935669,
5
+ "out": "out/glu-silu-waleed10-100L_run",
6
+ "run_name": "LM-glu-silu-waleed10-100L-16.9M-20260812-222757"
7
+ },
8
+ {
9
+ "variant": "glu-situglu-100L",
10
+ "eval_loss": 2.1253936290740967,
11
+ "out": "out/glu-situglu-100L_run",
12
+ "run_name": "LM-glu-situglu-100L-16.9M-20260812-224110"
13
+ },
14
+ {
15
+ "variant": "glu-waleed-100L",
16
+ "eval_loss": 2.1097662448883057,
17
+ "out": "out/glu-waleed-100L_run",
18
+ "run_name": "LM-glu-waleed-100L-16.9M-20260812-225525"
19
+ },
20
+ {
21
+ "variant": "glu-situglu_low-100L",
22
+ "eval_loss": 2.1262547969818115,
23
+ "out": "out/glu-situglu_low-100L_run",
24
+ "run_name": "LM-glu-situglu_low-100L-16.9M-20260812-230914"
25
+ },
26
+ {
27
+ "variant": "glu-waleedglu_low-100L",
28
+ "eval_loss": 2.106821060180664,
29
+ "out": "out/glu-waleedglu_low-100L_run",
30
+ "run_name": "LM-glu-waleedglu_low-100L-16.9M-20260812-232325"
31
+ },
32
+ {
33
+ "variant": "glu-linear-100L",
34
+ "eval_loss": 2.1097066402435303,
35
+ "out": "out/glu-linear-100L_run",
36
+ "run_name": "LM-glu-linear-100L-16.9M-20260812-233703"
37
+ }
38
+ ]
zain/Activation/wandb/debug-internal.log CHANGED
@@ -85,3 +85,35 @@
85
  {"time":"2026-08-12T23:46:35.998150976Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
86
  {"time":"2026-08-12T23:46:50.874132215Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":98,"history_lines":3,"events_offset":76,"events_lines":2,"console_offset":190,"console_lines":1}
87
  {"time":"2026-08-12T23:46:51.1109748Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
85
  {"time":"2026-08-12T23:46:35.998150976Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
86
  {"time":"2026-08-12T23:46:50.874132215Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":98,"history_lines":3,"events_offset":76,"events_lines":2,"console_offset":190,"console_lines":1}
87
  {"time":"2026-08-12T23:46:51.1109748Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
88
+ {"time":"2026-08-12T23:47:05.873911748Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":101,"history_lines":3,"events_offset":78,"events_lines":2,"console_offset":194,"console_lines":11}
89
+ {"time":"2026-08-12T23:47:06.048101755Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
90
+ {"time":"2026-08-12T23:47:20.874022923Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":104,"history_lines":2,"events_offset":80,"events_lines":2,"console_offset":200,"console_lines":1}
91
+ {"time":"2026-08-12T23:47:21.005671585Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
92
+ {"time":"2026-08-12T23:47:35.873635563Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":106,"history_lines":3,"events_offset":82,"events_lines":2,"console_offset":205,"console_lines":10}
93
+ {"time":"2026-08-12T23:47:36.000550646Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
94
+ {"time":"2026-08-12T23:47:50.87376451Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":109,"history_lines":2,"events_offset":84,"events_lines":2,"console_offset":210,"console_lines":1}
95
+ {"time":"2026-08-12T23:47:51.121247495Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
96
+ {"time":"2026-08-12T23:48:05.873619511Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":111,"history_lines":3,"events_offset":86,"events_lines":2,"console_offset":215,"console_lines":10}
97
+ {"time":"2026-08-12T23:48:06.082648628Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
98
+ {"time":"2026-08-12T23:48:20.874292434Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":114,"history_lines":2,"events_offset":88,"events_lines":2,"console_offset":220,"console_lines":1}
99
+ {"time":"2026-08-12T23:48:21.069644658Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
100
+ {"time":"2026-08-12T23:48:35.874379741Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":116,"history_lines":3,"events_offset":90,"events_lines":2,"console_offset":225,"console_lines":10}
101
+ {"time":"2026-08-12T23:48:36.100237591Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
102
+ {"time":"2026-08-12T23:48:50.873603862Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":119,"history_lines":2,"events_offset":92,"events_lines":2,"console_offset":230,"console_lines":1}
103
+ {"time":"2026-08-12T23:48:50.998416569Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
104
+ {"time":"2026-08-12T23:49:05.873747271Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":121,"history_lines":3,"events_offset":94,"events_lines":2,"console_offset":235,"console_lines":10}
105
+ {"time":"2026-08-12T23:49:06.067209608Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
106
+ {"time":"2026-08-12T23:49:20.873615557Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":96,"events_lines":2,"console_offset":240,"console_lines":1}
107
+ {"time":"2026-08-12T23:49:21.08163947Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
108
+ {"time":"2026-08-12T23:49:35.873636608Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":124,"history_lines":2,"events_offset":98,"events_lines":2,"console_offset":240,"console_lines":1}
109
+ {"time":"2026-08-12T23:49:35.988320548Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
110
+ {"time":"2026-08-12T23:49:50.873659846Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":126,"history_lines":2,"events_offset":100,"events_lines":2,"console_offset":240,"console_lines":1}
111
+ {"time":"2026-08-12T23:49:51.048489153Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
112
+ {"time":"2026-08-12T23:49:56.598587763Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
113
+ {"time":"2026-08-12T23:49:56.598789418Z","level":"INFO","msg":"filestream: sending request","total_files":3,"history_offset":128,"history_lines":1,"console_offset":245,"console_lines":15,"uploaded_len":3,"complete":true,"exit_code":0}
114
+ {"time":"2026-08-12T23:49:56.701784518Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
115
+ {"time":"2026-08-12T23:49:56.702912833Z","level":"INFO","msg":"handler: operation stats","stats":{}}
116
+ {"time":"2026-08-12T23:49:56.705484814Z","level":"INFO","msg":"stream: finishing up"}
117
+ {"time":"2026-08-12T23:49:56.705519356Z","level":"INFO","msg":"handler: closed"}
118
+ {"time":"2026-08-12T23:49:56.705632565Z","level":"INFO","msg":"sender: closed"}
119
+ {"time":"2026-08-12T23:49:56.705637913Z","level":"INFO","msg":"stream: all finished"}
zain/Activation/wandb/debug.log CHANGED
@@ -18,3 +18,8 @@ config: {'_wandb': {}}
18
  2026-08-12 23:37:05,310 INFO MainThread:3978000 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 100, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'glu', 'activation': 'linear', 'waleed_beta': 10.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/glu-linear-100L_run', 'per_device_train_batch_size': 64, 'num_train_epochs': 1, 'max_steps': 2500, 'learning_rate': 0.0007, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 500, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-glu-linear-100L-16.9M-20260812-233703', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 2498, 'eval_delay': 0, 'per_device_eval_batch_size': 1024, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-glu-linear-100L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
19
  2026-08-12 23:37:05,315 INFO MainThread:3978000 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 16934016 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x14d9ca094510>>
20
  2026-08-12 23:37:05,315 INFO MainThread:3978000 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 16934016 None
 
 
 
 
 
 
18
  2026-08-12 23:37:05,310 INFO MainThread:3978000 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 100, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'glu', 'activation': 'linear', 'waleed_beta': 10.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/glu-linear-100L_run', 'per_device_train_batch_size': 64, 'num_train_epochs': 1, 'max_steps': 2500, 'learning_rate': 0.0007, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 500, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-glu-linear-100L-16.9M-20260812-233703', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 2498, 'eval_delay': 0, 'per_device_eval_batch_size': 1024, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-glu-linear-100L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
19
  2026-08-12 23:37:05,315 INFO MainThread:3978000 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 16934016 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x14d9ca094510>>
20
  2026-08-12 23:37:05,315 INFO MainThread:3978000 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 16934016 None
21
+ 2026-08-12 23:49:56,215 INFO MainThread:3978000 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/research3/tpo5j00e
22
+ 2026-08-12 23:49:56,216 INFO MainThread:3978000 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0
23
+ 2026-08-12 23:49:56,216 INFO MainThread:3978000 [wandb_run.py:_restore():2570] restore
24
+ 2026-08-12 23:49:56,216 INFO MainThread:3978000 [wandb_run.py:_restore():2576] restore done
25
+ 2026-08-12 23:49:56,704 INFO MainThread:3978000 [wandb_run.py:_footer_sync_info():3993] logging synced files
zain/Activation/wandb/run-20260812_213354-drnrn7sc/logs/debug-core.log CHANGED
@@ -90,3 +90,11 @@
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
 
 
 
 
 
 
 
 
 
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
93
+ {"time":"2026-08-12T23:49:56.216535938Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
94
+ {"time":"2026-08-12T23:49:56.703966372Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
95
+ {"time":"2026-08-12T23:49:56.705445032Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"tpo5j00e","id":"5(@)"}
96
+ {"time":"2026-08-12T23:49:56.705984097Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"tpo5j00e","id":"5(@)"}
97
+ {"time":"2026-08-12T23:49:58.200430345Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
98
+ {"time":"2026-08-12T23:49:58.200526275Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
99
+ {"time":"2026-08-12T23:49:58.200433558Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
100
+ {"time":"2026-08-12T23:49:58.200535053Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
zain/Activation/wandb/run-20260812_213440-m4b216ej/logs/debug-core.log CHANGED
@@ -90,3 +90,11 @@
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
 
 
 
 
 
 
 
 
 
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
93
+ {"time":"2026-08-12T23:49:56.216535938Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
94
+ {"time":"2026-08-12T23:49:56.703966372Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
95
+ {"time":"2026-08-12T23:49:56.705445032Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"tpo5j00e","id":"5(@)"}
96
+ {"time":"2026-08-12T23:49:56.705984097Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"tpo5j00e","id":"5(@)"}
97
+ {"time":"2026-08-12T23:49:58.200430345Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
98
+ {"time":"2026-08-12T23:49:58.200526275Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
99
+ {"time":"2026-08-12T23:49:58.200433558Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
100
+ {"time":"2026-08-12T23:49:58.200535053Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
zain/Activation/wandb/run-20260812_213624-1sryr5l2/logs/debug-core.log CHANGED
@@ -90,3 +90,11 @@
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
 
 
 
 
 
 
 
 
 
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
93
+ {"time":"2026-08-12T23:49:56.216535938Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
94
+ {"time":"2026-08-12T23:49:56.703966372Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
95
+ {"time":"2026-08-12T23:49:56.705445032Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"tpo5j00e","id":"5(@)"}
96
+ {"time":"2026-08-12T23:49:56.705984097Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"tpo5j00e","id":"5(@)"}
97
+ {"time":"2026-08-12T23:49:58.200430345Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
98
+ {"time":"2026-08-12T23:49:58.200526275Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
99
+ {"time":"2026-08-12T23:49:58.200433558Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
100
+ {"time":"2026-08-12T23:49:58.200535053Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
zain/Activation/wandb/run-20260812_220132-25l266uo/logs/debug-core.log CHANGED
@@ -90,3 +90,11 @@
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
 
 
 
 
 
 
 
 
 
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
93
+ {"time":"2026-08-12T23:49:56.216535938Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
94
+ {"time":"2026-08-12T23:49:56.703966372Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
95
+ {"time":"2026-08-12T23:49:56.705445032Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"tpo5j00e","id":"5(@)"}
96
+ {"time":"2026-08-12T23:49:56.705984097Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"tpo5j00e","id":"5(@)"}
97
+ {"time":"2026-08-12T23:49:58.200430345Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
98
+ {"time":"2026-08-12T23:49:58.200526275Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
99
+ {"time":"2026-08-12T23:49:58.200433558Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
100
+ {"time":"2026-08-12T23:49:58.200535053Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
zain/Activation/wandb/run-20260812_222706-8exldg55/logs/debug-core.log CHANGED
@@ -90,3 +90,11 @@
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
 
 
 
 
 
 
 
 
 
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
93
+ {"time":"2026-08-12T23:49:56.216535938Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
94
+ {"time":"2026-08-12T23:49:56.703966372Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
95
+ {"time":"2026-08-12T23:49:56.705445032Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"tpo5j00e","id":"5(@)"}
96
+ {"time":"2026-08-12T23:49:56.705984097Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"tpo5j00e","id":"5(@)"}
97
+ {"time":"2026-08-12T23:49:58.200430345Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
98
+ {"time":"2026-08-12T23:49:58.200526275Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
99
+ {"time":"2026-08-12T23:49:58.200433558Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
100
+ {"time":"2026-08-12T23:49:58.200535053Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
zain/Activation/wandb/run-20260812_222758-48ajimfv/logs/debug-core.log CHANGED
@@ -90,3 +90,11 @@
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
 
 
 
 
 
 
 
 
 
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
93
+ {"time":"2026-08-12T23:49:56.216535938Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
94
+ {"time":"2026-08-12T23:49:56.703966372Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
95
+ {"time":"2026-08-12T23:49:56.705445032Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"tpo5j00e","id":"5(@)"}
96
+ {"time":"2026-08-12T23:49:56.705984097Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"tpo5j00e","id":"5(@)"}
97
+ {"time":"2026-08-12T23:49:58.200430345Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
98
+ {"time":"2026-08-12T23:49:58.200526275Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
99
+ {"time":"2026-08-12T23:49:58.200433558Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
100
+ {"time":"2026-08-12T23:49:58.200535053Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
zain/Activation/wandb/run-20260812_224111-g6mesp08/logs/debug-core.log CHANGED
@@ -90,3 +90,11 @@
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
 
 
 
 
 
 
 
 
 
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
93
+ {"time":"2026-08-12T23:49:56.216535938Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
94
+ {"time":"2026-08-12T23:49:56.703966372Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
95
+ {"time":"2026-08-12T23:49:56.705445032Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"tpo5j00e","id":"5(@)"}
96
+ {"time":"2026-08-12T23:49:56.705984097Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"tpo5j00e","id":"5(@)"}
97
+ {"time":"2026-08-12T23:49:58.200430345Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
98
+ {"time":"2026-08-12T23:49:58.200526275Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
99
+ {"time":"2026-08-12T23:49:58.200433558Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
100
+ {"time":"2026-08-12T23:49:58.200535053Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
zain/Activation/wandb/run-20260812_225525-hf77resg/logs/debug-core.log CHANGED
@@ -90,3 +90,11 @@
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
 
 
 
 
 
 
 
 
 
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
93
+ {"time":"2026-08-12T23:49:56.216535938Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
94
+ {"time":"2026-08-12T23:49:56.703966372Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
95
+ {"time":"2026-08-12T23:49:56.705445032Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"tpo5j00e","id":"5(@)"}
96
+ {"time":"2026-08-12T23:49:56.705984097Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"tpo5j00e","id":"5(@)"}
97
+ {"time":"2026-08-12T23:49:58.200430345Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
98
+ {"time":"2026-08-12T23:49:58.200526275Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
99
+ {"time":"2026-08-12T23:49:58.200433558Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
100
+ {"time":"2026-08-12T23:49:58.200535053Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
zain/Activation/wandb/run-20260812_230915-vvjf0upl/logs/debug-core.log CHANGED
@@ -90,3 +90,11 @@
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
 
 
 
 
 
 
 
 
 
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
93
+ {"time":"2026-08-12T23:49:56.216535938Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
94
+ {"time":"2026-08-12T23:49:56.703966372Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
95
+ {"time":"2026-08-12T23:49:56.705445032Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"tpo5j00e","id":"5(@)"}
96
+ {"time":"2026-08-12T23:49:56.705984097Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"tpo5j00e","id":"5(@)"}
97
+ {"time":"2026-08-12T23:49:58.200430345Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
98
+ {"time":"2026-08-12T23:49:58.200526275Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
99
+ {"time":"2026-08-12T23:49:58.200433558Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
100
+ {"time":"2026-08-12T23:49:58.200535053Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
zain/Activation/wandb/run-20260812_232326-8u0k1nti/logs/debug-core.log CHANGED
@@ -90,3 +90,11 @@
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
 
 
 
 
 
 
 
 
 
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
93
+ {"time":"2026-08-12T23:49:56.216535938Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
94
+ {"time":"2026-08-12T23:49:56.703966372Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
95
+ {"time":"2026-08-12T23:49:56.705445032Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"tpo5j00e","id":"5(@)"}
96
+ {"time":"2026-08-12T23:49:56.705984097Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"tpo5j00e","id":"5(@)"}
97
+ {"time":"2026-08-12T23:49:58.200430345Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
98
+ {"time":"2026-08-12T23:49:58.200526275Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
99
+ {"time":"2026-08-12T23:49:58.200433558Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
100
+ {"time":"2026-08-12T23:49:58.200535053Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
zain/Activation/wandb/run-20260812_233704-tpo5j00e/files/config.yaml ADDED
@@ -0,0 +1,438 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _name_or_path:
2
+ value: ""
3
+ _wandb:
4
+ value:
5
+ cli_version: 0.28.1
6
+ e:
7
+ 4qopxlj5t12670m45s6hbdlio87c05xg:
8
+ args:
9
+ - --config
10
+ - /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/configs/baseline100L.yaml
11
+ - --variants
12
+ - glu-silu-waleed10
13
+ - glu-situglu
14
+ - glu-waleed
15
+ - glu-situglu_low
16
+ - glu-waleedglu_low
17
+ - glu-linear
18
+ codePath: sweep.py
19
+ codePathLocal: sweep.py
20
+ cpu_count: 112
21
+ cpu_count_logical: 224
22
+ cudaVersion: "12.4"
23
+ disk:
24
+ /:
25
+ total: "1560765693952"
26
+ used: "716448854016"
27
+ email: deepnevro@gmail.com
28
+ executable: /mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python
29
+ git:
30
+ commit: 26a43c7dee7982e610a25ac78443de9ce18f5077
31
+ remote: https://github.com/w-ahmad1a10/Activation.git
32
+ gpu: NVIDIA H100 80GB HBM3
33
+ gpu_count: 8
34
+ gpu_nvidia:
35
+ - architecture: Hopper
36
+ cudaCores: 16896
37
+ memoryTotal: "85520809984"
38
+ name: NVIDIA H100 80GB HBM3
39
+ uuid: GPU-39c684a5-fde6-83d7-1663-0859795881ae
40
+ - architecture: Hopper
41
+ cudaCores: 16896
42
+ memoryTotal: "85520809984"
43
+ name: NVIDIA H100 80GB HBM3
44
+ uuid: GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3
45
+ - architecture: Hopper
46
+ cudaCores: 16896
47
+ memoryTotal: "85520809984"
48
+ name: NVIDIA H100 80GB HBM3
49
+ uuid: GPU-132944c4-b689-2b5f-89a4-d730401677ab
50
+ - architecture: Hopper
51
+ cudaCores: 16896
52
+ memoryTotal: "85520809984"
53
+ name: NVIDIA H100 80GB HBM3
54
+ uuid: GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864
55
+ - architecture: Hopper
56
+ cudaCores: 16896
57
+ memoryTotal: "85520809984"
58
+ name: NVIDIA H100 80GB HBM3
59
+ uuid: GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef
60
+ - architecture: Hopper
61
+ cudaCores: 16896
62
+ memoryTotal: "85520809984"
63
+ name: NVIDIA H100 80GB HBM3
64
+ uuid: GPU-bc6c3e3c-9b90-09ca-c034-774961847c54
65
+ - architecture: Hopper
66
+ cudaCores: 16896
67
+ memoryTotal: "85520809984"
68
+ name: NVIDIA H100 80GB HBM3
69
+ uuid: GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9
70
+ - architecture: Hopper
71
+ cudaCores: 16896
72
+ memoryTotal: "85520809984"
73
+ name: NVIDIA H100 80GB HBM3
74
+ uuid: GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea
75
+ host: deeplens-k3s-node1
76
+ memory:
77
+ total: "2164089937920"
78
+ os: Linux-5.15.0-126-generic-x86_64-with-glibc2.35
79
+ program: /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/sweep.py
80
+ python: CPython 3.11.15
81
+ root: /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation
82
+ startedAt: "2026-08-12T23:37:04.657044Z"
83
+ writerId: 4qopxlj5t12670m45s6hbdlio87c05xg
84
+ m:
85
+ - "1": train/global_step
86
+ "6":
87
+ - 3
88
+ "7": []
89
+ - "2": '*'
90
+ "5": 1
91
+ "6":
92
+ - 1
93
+ "7": []
94
+ python_version: 3.11.15
95
+ t:
96
+ "1":
97
+ - 1
98
+ - 5
99
+ - 11
100
+ - 41
101
+ - 49
102
+ - 51
103
+ - 53
104
+ - 71
105
+ "2":
106
+ - 1
107
+ - 5
108
+ - 11
109
+ - 41
110
+ - 49
111
+ - 51
112
+ - 53
113
+ - 71
114
+ "3":
115
+ - 2
116
+ - 7
117
+ - 13
118
+ - 19
119
+ - 62
120
+ - 66
121
+ "4": 3.11.15
122
+ "5": 0.28.1
123
+ "6": 5.16.0.dev0
124
+ "9":
125
+ "1": transformers_trainer
126
+ "12": 0.28.1
127
+ "13": linux-x86_64
128
+ accelerator_config:
129
+ value:
130
+ dispatch_batches: null
131
+ even_batches: true
132
+ gradient_accumulation_kwargs: null
133
+ non_blocking: false
134
+ split_batches: false
135
+ use_seedable_sampler: true
136
+ activation:
137
+ value: linear
138
+ adam_beta1:
139
+ value: 0.9
140
+ adam_beta2:
141
+ value: 0.999
142
+ adam_epsilon:
143
+ value: 1e-08
144
+ architectures:
145
+ value: null
146
+ attention_bias:
147
+ value: false
148
+ attention_dropout:
149
+ value: 0
150
+ auto_find_batch_size:
151
+ value: false
152
+ average_tokens_across_devices:
153
+ value: true
154
+ batch_eval_metrics:
155
+ value: false
156
+ bf16:
157
+ value: true
158
+ bf16_full_eval:
159
+ value: false
160
+ bos_token_id:
161
+ value: 1
162
+ chunk_size_feed_forward:
163
+ value: 0
164
+ data_seed:
165
+ value: 42
166
+ dataloader_drop_last:
167
+ value: false
168
+ dataloader_in_order:
169
+ value: true
170
+ dataloader_multiprocessing_context:
171
+ value: null
172
+ dataloader_num_workers:
173
+ value: 0
174
+ dataloader_persistent_workers:
175
+ value: false
176
+ dataloader_pin_memory:
177
+ value: true
178
+ dataloader_prefetch_factor:
179
+ value: null
180
+ ddp_backend:
181
+ value: null
182
+ ddp_broadcast_buffers:
183
+ value: null
184
+ ddp_bucket_cap_mb:
185
+ value: null
186
+ ddp_find_unused_parameters:
187
+ value: null
188
+ ddp_static_graph:
189
+ value: null
190
+ ddp_timeout:
191
+ value: 1800
192
+ debug:
193
+ value: []
194
+ deepspeed:
195
+ value: null
196
+ disable_tqdm:
197
+ value: false
198
+ do_eval:
199
+ value: true
200
+ do_predict:
201
+ value: false
202
+ do_train:
203
+ value: false
204
+ dtype:
205
+ value: null
206
+ enable_jit_checkpoint:
207
+ value: false
208
+ eos_token_id:
209
+ value: 2
210
+ eval_accumulation_steps:
211
+ value: null
212
+ eval_delay:
213
+ value: 0
214
+ eval_do_concat_batches:
215
+ value: true
216
+ eval_on_start:
217
+ value: false
218
+ eval_steps:
219
+ value: 2498
220
+ eval_strategy:
221
+ value: steps
222
+ eval_use_gather_object:
223
+ value: false
224
+ fp16:
225
+ value: false
226
+ fp16_full_eval:
227
+ value: false
228
+ fsdp:
229
+ value: null
230
+ fsdp_config:
231
+ value: null
232
+ full_determinism:
233
+ value: false
234
+ gradient_accumulation_steps:
235
+ value: 1
236
+ gradient_checkpointing:
237
+ value: false
238
+ gradient_checkpointing_kwargs:
239
+ value: null
240
+ greater_is_better:
241
+ value: null
242
+ head_dim:
243
+ value: 32
244
+ hidden_act:
245
+ value: silu
246
+ hidden_size:
247
+ value: 128
248
+ hub_always_push:
249
+ value: false
250
+ hub_model_id:
251
+ value: w-ahmad/6L-glu-linear-100L
252
+ hub_private_repo:
253
+ value: null
254
+ hub_revision:
255
+ value: null
256
+ hub_strategy:
257
+ value: every_save
258
+ hub_token:
259
+ value: <HUB_TOKEN>
260
+ id2label:
261
+ value:
262
+ "0": LABEL_0
263
+ "1": LABEL_1
264
+ ignore_data_skip:
265
+ value: false
266
+ include_for_metrics:
267
+ value: []
268
+ include_num_input_tokens_seen:
269
+ value: "no"
270
+ initializer_range:
271
+ value: 0.02
272
+ intermediate_size:
273
+ value: 256
274
+ is_encoder_decoder:
275
+ value: false
276
+ label_names:
277
+ value: null
278
+ label_smoothing_factor:
279
+ value: 0
280
+ label2id:
281
+ value:
282
+ LABEL_0: 0
283
+ LABEL_1: 1
284
+ learning_rate:
285
+ value: 0.0007
286
+ length_column_name:
287
+ value: length
288
+ liger_kernel_config:
289
+ value: null
290
+ load_best_model_at_end:
291
+ value: false
292
+ local_rank:
293
+ value: -1
294
+ log_level:
295
+ value: passive
296
+ log_level_replica:
297
+ value: warning
298
+ log_on_each_node:
299
+ value: true
300
+ logging_first_step:
301
+ value: false
302
+ logging_nan_inf_filter:
303
+ value: true
304
+ logging_steps:
305
+ value: 20
306
+ logging_strategy:
307
+ value: steps
308
+ lr_scheduler_kwargs:
309
+ value: null
310
+ lr_scheduler_type:
311
+ value: constant_with_warmup
312
+ max_grad_norm:
313
+ value: 1
314
+ max_position_embeddings:
315
+ value: 512
316
+ max_steps:
317
+ value: 2500
318
+ metric_for_best_model:
319
+ value: null
320
+ mlp_bias:
321
+ value: false
322
+ mlp_type:
323
+ value: glu
324
+ model/num_parameters:
325
+ value: 16934016
326
+ model_type:
327
+ value: tiny_llama
328
+ neftune_noise_alpha:
329
+ value: null
330
+ num_attention_heads:
331
+ value: 4
332
+ num_hidden_layers:
333
+ value: 100
334
+ num_key_value_heads:
335
+ value: 4
336
+ num_train_epochs:
337
+ value: 1
338
+ optim:
339
+ value: adamw_torch_fused
340
+ optim_args:
341
+ value: null
342
+ optim_target_modules:
343
+ value: null
344
+ output_attentions:
345
+ value: false
346
+ output_dir:
347
+ value: out/glu-linear-100L_run
348
+ output_hidden_states:
349
+ value: false
350
+ pad_token_id:
351
+ value: 0
352
+ parallelism_config:
353
+ value: null
354
+ per_device_eval_batch_size:
355
+ value: 1024
356
+ per_device_train_batch_size:
357
+ value: 64
358
+ prediction_loss_only:
359
+ value: false
360
+ pretraining_tp:
361
+ value: 1
362
+ problem_type:
363
+ value: null
364
+ project:
365
+ value: huggingface
366
+ push_to_hub:
367
+ value: false
368
+ remove_unused_columns:
369
+ value: false
370
+ report_to:
371
+ value:
372
+ - wandb
373
+ restore_callback_states_from_checkpoint:
374
+ value: false
375
+ resume_from_checkpoint:
376
+ value: null
377
+ return_dict:
378
+ value: true
379
+ rms_norm_eps:
380
+ value: 1e-06
381
+ rope_parameters:
382
+ value:
383
+ rope_theta: 10000
384
+ rope_type: default
385
+ run_name:
386
+ value: LM-glu-linear-100L-16.9M-20260812-233703
387
+ save_on_each_node:
388
+ value: false
389
+ save_only_model:
390
+ value: false
391
+ save_steps:
392
+ value: 100
393
+ save_strategy:
394
+ value: steps
395
+ save_total_limit:
396
+ value: null
397
+ seed:
398
+ value: 42
399
+ skip_memory_metrics:
400
+ value: true
401
+ tf32:
402
+ value: null
403
+ tie_word_embeddings:
404
+ value: true
405
+ tokenizer_name:
406
+ value: w-ahmad/tiny-stories-tokenizer
407
+ torch_compile:
408
+ value: false
409
+ torch_compile_backend:
410
+ value: null
411
+ torch_compile_mode:
412
+ value: null
413
+ torch_empty_cache_steps:
414
+ value: null
415
+ trackio_bucket_id:
416
+ value: null
417
+ trackio_space_id:
418
+ value: null
419
+ trackio_static_space_id:
420
+ value: null
421
+ train_sampling_strategy:
422
+ value: random
423
+ transformers_version:
424
+ value: 5.16.0.dev0
425
+ use_cache:
426
+ value: false
427
+ use_cpu:
428
+ value: false
429
+ use_liger_kernel:
430
+ value: false
431
+ vocab_size:
432
+ value: 4096
433
+ waleed_beta:
434
+ value: 10
435
+ warmup_steps:
436
+ value: 500
437
+ weight_decay:
438
+ value: 0
zain/Activation/wandb/run-20260812_233704-tpo5j00e/files/output.log CHANGED
@@ -198,5 +198,63 @@ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 19.
198
  - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
199
  - If you are not the owner of the model architecture class, please contact the model code owner to update it.
200
  Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 18.11it/s]
201
- 81%|████████ | 2029/2500 [09:46<02:15, 3.48it/s], ?it/s]
202
  {'loss': '2.191', 'grad_norm': '0.4023', 'learning_rate': '0.0007', 'epoch': '0.1362', 'train/total_time_seconds': '476.3', 'train/time_per_step_avg': '0.2381', 'train/epoch_time_elapsed': '582.7', 'train/estimated_remaining_minutes': '1.886'}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
198
  - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
199
  - If you are not the owner of the model architecture class, please contact the model code owner to update it.
200
  Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 18.11it/s]
201
+ 84%|████████| 2100/2500 [10:06<01:57, 3.39it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
202
  {'loss': '2.191', 'grad_norm': '0.4023', 'learning_rate': '0.0007', 'epoch': '0.1362', 'train/total_time_seconds': '476.3', 'train/time_per_step_avg': '0.2381', 'train/epoch_time_elapsed': '582.7', 'train/estimated_remaining_minutes': '1.886'}
203
+ {'loss': '2.171', 'grad_norm': '0.4434', 'learning_rate': '0.0007', 'epoch': '0.1375', 'train/total_time_seconds': '481', 'train/time_per_step_avg': '0.2381', 'train/epoch_time_elapsed': '588.4', 'train/estimated_remaining_minutes': '1.808'}
204
+ {'loss': '2.189', 'grad_norm': '0.3965', 'learning_rate': '0.0007', 'epoch': '0.1389', 'train/total_time_seconds': '485.7', 'train/time_per_step_avg': '0.2382', 'train/epoch_time_elapsed': '594.1', 'train/estimated_remaining_minutes': '1.729'}
205
+ {'loss': '2.167', 'grad_norm': '0.3887', 'learning_rate': '0.0007', 'epoch': '0.1402', 'train/total_time_seconds': '490.5', 'train/time_per_step_avg': '0.2383', 'train/epoch_time_elapsed': '599.9', 'train/estimated_remaining_minutes': '1.651'}
206
+ {'loss': '2.157', 'grad_norm': '0.377', 'learning_rate': '0.0007', 'epoch': '0.1416', 'train/total_time_seconds': '495.4', 'train/time_per_step_avg': '0.2406', 'train/epoch_time_elapsed': '605.9', 'train/estimated_remaining_minutes': '1.573'}
207
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
208
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
209
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
210
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 14.67it/s]
211
+ 88%|████████▊ | 2200/2500 [10:36<01:27, 3.41it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
212
+ {'loss': '2.166', 'grad_norm': '0.3828', 'learning_rate': '0.0007', 'epoch': '0.1429', 'train/total_time_seconds': '500.2', 'train/time_per_step_avg': '0.2388', 'train/epoch_time_elapsed': '612', 'train/estimated_remaining_minutes': '1.494'}
213
+ {'loss': '2.167', 'grad_norm': '0.4023', 'learning_rate': '0.0007', 'epoch': '0.1443', 'train/total_time_seconds': '505.1', 'train/time_per_step_avg': '0.2405', 'train/epoch_time_elapsed': '618', 'train/estimated_remaining_minutes': '1.416'}
214
+ {'loss': '2.152', 'grad_norm': '0.3887', 'learning_rate': '0.0007', 'epoch': '0.1456', 'train/total_time_seconds': '509.9', 'train/time_per_step_avg': '0.2419', 'train/epoch_time_elapsed': '623.8', 'train/estimated_remaining_minutes': '1.338'}
215
+ {'loss': '2.162', 'grad_norm': '0.3945', 'learning_rate': '0.0007', 'epoch': '0.1469', 'train/total_time_seconds': '514.7', 'train/time_per_step_avg': '0.2424', 'train/epoch_time_elapsed': '629.6', 'train/estimated_remaining_minutes': '1.259'}
216
+ {'loss': '2.136', 'grad_norm': '0.3848', 'learning_rate': '0.0007', 'epoch': '0.1483', 'train/total_time_seconds': '519.5', 'train/time_per_step_avg': '0.2416', 'train/epoch_time_elapsed': '635.5', 'train/estimated_remaining_minutes': '1.181'}
217
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
218
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
219
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
220
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 16.67it/s]
221
+ 92%|█████████▏| 2300/2500 [11:05<01:00, 3.28it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
222
+ {'loss': '2.144', 'grad_norm': '0.4062', 'learning_rate': '0.0007', 'epoch': '0.1496', 'train/total_time_seconds': '524.4', 'train/time_per_step_avg': '0.2416', 'train/epoch_time_elapsed': '641.7', 'train/estimated_remaining_minutes': '1.102'}
223
+ {'loss': '2.156', 'grad_norm': '0.4199', 'learning_rate': '0.0007', 'epoch': '0.151', 'train/total_time_seconds': '529.2', 'train/time_per_step_avg': '0.241', 'train/epoch_time_elapsed': '647.5', 'train/estimated_remaining_minutes': '1.024'}
224
+ {'loss': '2.133', 'grad_norm': '0.4043', 'learning_rate': '0.0007', 'epoch': '0.1523', 'train/total_time_seconds': '534.1', 'train/time_per_step_avg': '0.2416', 'train/epoch_time_elapsed': '653.4', 'train/estimated_remaining_minutes': '0.9452'}
225
+ {'loss': '2.144', 'grad_norm': '0.3789', 'learning_rate': '0.0007', 'epoch': '0.1537', 'train/total_time_seconds': '538.9', 'train/time_per_step_avg': '0.2421', 'train/epoch_time_elapsed': '659.3', 'train/estimated_remaining_minutes': '0.8667'}
226
+ {'loss': '2.127', 'grad_norm': '0.3945', 'learning_rate': '0.0007', 'epoch': '0.155', 'train/total_time_seconds': '543.8', 'train/time_per_step_avg': '0.2423', 'train/epoch_time_elapsed': '665.2', 'train/estimated_remaining_minutes': '0.7881'}
227
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
228
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
229
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
230
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 17.38it/s]
231
+ 96%|█████████▌| 2400/2500 [11:35<00:29, 3.38it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
232
+ {'loss': '2.135', 'grad_norm': '0.4121', 'learning_rate': '0.0007', 'epoch': '0.1564', 'train/total_time_seconds': '548.8', 'train/time_per_step_avg': '0.2445', 'train/epoch_time_elapsed': '671.6', 'train/estimated_remaining_minutes': '0.7097'}
233
+ {'loss': '2.124', 'grad_norm': '0.3711', 'learning_rate': '0.0007', 'epoch': '0.1577', 'train/total_time_seconds': '553.6', 'train/time_per_step_avg': '0.2445', 'train/epoch_time_elapsed': '677.4', 'train/estimated_remaining_minutes': '0.6309'}
234
+ {'loss': '2.127', 'grad_norm': '0.3867', 'learning_rate': '0.0007', 'epoch': '0.1591', 'train/total_time_seconds': '558.4', 'train/time_per_step_avg': '0.2438', 'train/epoch_time_elapsed': '683.2', 'train/estimated_remaining_minutes': '0.5521'}
235
+ {'loss': '2.13', 'grad_norm': '0.3887', 'learning_rate': '0.0007', 'epoch': '0.1604', 'train/total_time_seconds': '563.2', 'train/time_per_step_avg': '0.2433', 'train/epoch_time_elapsed': '689.1', 'train/estimated_remaining_minutes': '0.4733'}
236
+ {'loss': '2.116', 'grad_norm': '0.4062', 'learning_rate': '0.0007', 'epoch': '0.1618', 'train/total_time_seconds': '568.1', 'train/time_per_step_avg': '0.2433', 'train/epoch_time_elapsed': '695', 'train/estimated_remaining_minutes': '0.3945'}
237
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
238
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
239
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
240
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 16.05it/s]
241
+ 100%|██████████| 2500/2500 [12:35<00:00, 3.51s[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
242
+ {'loss': '2.118', 'grad_norm': '0.4023', 'learning_rate': '0.0007', 'epoch': '0.1631', 'train/total_time_seconds': '572.9', 'train/time_per_step_avg': '0.2413', 'train/epoch_time_elapsed': '701.1', 'train/estimated_remaining_minutes': '0.3157'}
243
+ {'loss': '2.105', 'grad_norm': '0.3809', 'learning_rate': '0.0007', 'epoch': '0.1645', 'train/total_time_seconds': '577.8', 'train/time_per_step_avg': '0.2416', 'train/epoch_time_elapsed': '707', 'train/estimated_remaining_minutes': '0.2368'}
244
+ {'loss': '2.117', 'grad_norm': '0.3828', 'learning_rate': '0.0007', 'epoch': '0.1658', 'train/total_time_seconds': '582.6', 'train/time_per_step_avg': '0.2415', 'train/epoch_time_elapsed': '712.8', 'train/estimated_remaining_minutes': '0.1579'}
245
+ {'loss': '2.117', 'grad_norm': '0.3926', 'learning_rate': '0.0007', 'epoch': '0.1672', 'train/total_time_seconds': '587.4', 'train/time_per_step_avg': '0.2417', 'train/epoch_time_elapsed': '718.7', 'train/estimated_remaining_minutes': '0.07895'}
246
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
247
+ {'eval_loss': '2.11', 'eval_runtime': '15.33', 'eval_samples_per_second': '621.6', 'eval_steps_per_second': '0.652', 'epoch': '0.1684', 'train/total_time_seconds': '591.8', 'train/time_per_step_avg': '0.2416', 'train/epoch_time_elapsed': '739.3', 'train/estimated_remaining_minutes': '0.007896'}
248
+ {'loss': '2.1', 'grad_norm': '0.3789', 'learning_rate': '0.0007', 'epoch': '0.1685', 'train/total_time_seconds': '592.2', 'train/time_per_step_avg': '0.2415', 'train/epoch_time_elapsed': '739.8', 'train/estimated_remaining_minutes': '0'}
249
+ {'eval_loss': '2.11', 'eval_runtime': '15.33', 'eval_samples_per_second': '621.4', 'eval_steps_per_second': '0.652', 'epoch': '0.1685', 'train/total_time_seconds': '592.2', 'train/time_per_step_avg': '0.2415', 'train/epoch_time_elapsed': '755.2', 'train/estimated_remaining_minutes': '0'}
250
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
251
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
252
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 13.65it/s]
253
+ 100%|██████████| 2500/2500 [12:36<00:00, 3.31it/s], ?it/s]
254
+ {'train_runtime': '756.2', 'train_samples_per_second': '211.6', 'train_steps_per_second': '3.306', 'train_loss': '3.001', 'epoch': '0.1685', 'train/total_time_seconds': '592.2', 'train/time_per_step_avg': '0.2415', 'train/epoch_time_elapsed': '755.5', 'train/estimated_remaining_minutes': '0'}
255
+ 100%|██████████| 10/10 [00:13<00:00, 1.31s/it]
256
+ [transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
257
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
258
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
259
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
260
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 17.20it/s]
zain/Activation/wandb/run-20260812_233704-tpo5j00e/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_wandb":{"runtime":770},"train/train/total_time_seconds":592.2469161488116,"train_runtime":756.2272,"eval/runtime":15.3092,"train/train/epoch_time_elapsed":770.8056703023612,"train_samples_per_second":211.577,"total_flos":8.06570950656e+15,"_runtime":770,"_step":128,"eval/loss":2.1097066402435303,"train_steps_per_second":3.306,"eval/samples_per_second":622.305,"train/train/time_per_step_avg":0.2415289905667305,"train/grad_norm":0.37890625,"eval/steps_per_second":0.653,"_timestamp":1.7865785961217074e+09,"train/train/estimated_remaining_minutes":0,"train/learning_rate":0.0007,"train/loss":2.100436973571777,"train/global_step":2500,"train_loss":3.001207452392578,"train/epoch":0.16852039096730703}
zain/Activation/wandb/run-20260812_233704-tpo5j00e/logs/debug-core.log CHANGED
@@ -90,3 +90,11 @@
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
 
 
 
 
 
 
 
 
 
90
  {"time":"2026-08-12T23:37:04.659365837Z","level":"INFO","msg":"handleInformInit: received","streamId":"tpo5j00e","id":"5(@)"}
91
  {"time":"2026-08-12T23:37:04.922876846Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"tpo5j00e","id":"5(@)"}
92
  {"time":"2026-08-12T23:37:10.31045638Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
93
+ {"time":"2026-08-12T23:49:56.216535938Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
94
+ {"time":"2026-08-12T23:49:56.703966372Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"zbcpym8ebxly"}
95
+ {"time":"2026-08-12T23:49:56.705445032Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"tpo5j00e","id":"5(@)"}
96
+ {"time":"2026-08-12T23:49:56.705984097Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"tpo5j00e","id":"5(@)"}
97
+ {"time":"2026-08-12T23:49:58.200430345Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
98
+ {"time":"2026-08-12T23:49:58.200526275Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
99
+ {"time":"2026-08-12T23:49:58.200433558Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
100
+ {"time":"2026-08-12T23:49:58.200535053Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
zain/Activation/wandb/run-20260812_233704-tpo5j00e/logs/debug-internal.log CHANGED
@@ -85,3 +85,35 @@
85
  {"time":"2026-08-12T23:46:35.998150976Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
86
  {"time":"2026-08-12T23:46:50.874132215Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":98,"history_lines":3,"events_offset":76,"events_lines":2,"console_offset":190,"console_lines":1}
87
  {"time":"2026-08-12T23:46:51.1109748Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
85
  {"time":"2026-08-12T23:46:35.998150976Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
86
  {"time":"2026-08-12T23:46:50.874132215Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":98,"history_lines":3,"events_offset":76,"events_lines":2,"console_offset":190,"console_lines":1}
87
  {"time":"2026-08-12T23:46:51.1109748Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
88
+ {"time":"2026-08-12T23:47:05.873911748Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":101,"history_lines":3,"events_offset":78,"events_lines":2,"console_offset":194,"console_lines":11}
89
+ {"time":"2026-08-12T23:47:06.048101755Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
90
+ {"time":"2026-08-12T23:47:20.874022923Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":104,"history_lines":2,"events_offset":80,"events_lines":2,"console_offset":200,"console_lines":1}
91
+ {"time":"2026-08-12T23:47:21.005671585Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
92
+ {"time":"2026-08-12T23:47:35.873635563Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":106,"history_lines":3,"events_offset":82,"events_lines":2,"console_offset":205,"console_lines":10}
93
+ {"time":"2026-08-12T23:47:36.000550646Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
94
+ {"time":"2026-08-12T23:47:50.87376451Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":109,"history_lines":2,"events_offset":84,"events_lines":2,"console_offset":210,"console_lines":1}
95
+ {"time":"2026-08-12T23:47:51.121247495Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
96
+ {"time":"2026-08-12T23:48:05.873619511Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":111,"history_lines":3,"events_offset":86,"events_lines":2,"console_offset":215,"console_lines":10}
97
+ {"time":"2026-08-12T23:48:06.082648628Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
98
+ {"time":"2026-08-12T23:48:20.874292434Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":114,"history_lines":2,"events_offset":88,"events_lines":2,"console_offset":220,"console_lines":1}
99
+ {"time":"2026-08-12T23:48:21.069644658Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
100
+ {"time":"2026-08-12T23:48:35.874379741Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":116,"history_lines":3,"events_offset":90,"events_lines":2,"console_offset":225,"console_lines":10}
101
+ {"time":"2026-08-12T23:48:36.100237591Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
102
+ {"time":"2026-08-12T23:48:50.873603862Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":119,"history_lines":2,"events_offset":92,"events_lines":2,"console_offset":230,"console_lines":1}
103
+ {"time":"2026-08-12T23:48:50.998416569Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
104
+ {"time":"2026-08-12T23:49:05.873747271Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":121,"history_lines":3,"events_offset":94,"events_lines":2,"console_offset":235,"console_lines":10}
105
+ {"time":"2026-08-12T23:49:06.067209608Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
106
+ {"time":"2026-08-12T23:49:20.873615557Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":96,"events_lines":2,"console_offset":240,"console_lines":1}
107
+ {"time":"2026-08-12T23:49:21.08163947Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
108
+ {"time":"2026-08-12T23:49:35.873636608Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":124,"history_lines":2,"events_offset":98,"events_lines":2,"console_offset":240,"console_lines":1}
109
+ {"time":"2026-08-12T23:49:35.988320548Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
110
+ {"time":"2026-08-12T23:49:50.873659846Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":126,"history_lines":2,"events_offset":100,"events_lines":2,"console_offset":240,"console_lines":1}
111
+ {"time":"2026-08-12T23:49:51.048489153Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
112
+ {"time":"2026-08-12T23:49:56.598587763Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
113
+ {"time":"2026-08-12T23:49:56.598789418Z","level":"INFO","msg":"filestream: sending request","total_files":3,"history_offset":128,"history_lines":1,"console_offset":245,"console_lines":15,"uploaded_len":3,"complete":true,"exit_code":0}
114
+ {"time":"2026-08-12T23:49:56.701784518Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
115
+ {"time":"2026-08-12T23:49:56.702912833Z","level":"INFO","msg":"handler: operation stats","stats":{}}
116
+ {"time":"2026-08-12T23:49:56.705484814Z","level":"INFO","msg":"stream: finishing up"}
117
+ {"time":"2026-08-12T23:49:56.705519356Z","level":"INFO","msg":"handler: closed"}
118
+ {"time":"2026-08-12T23:49:56.705632565Z","level":"INFO","msg":"sender: closed"}
119
+ {"time":"2026-08-12T23:49:56.705637913Z","level":"INFO","msg":"stream: all finished"}
zain/Activation/wandb/run-20260812_233704-tpo5j00e/logs/debug.log CHANGED
@@ -18,3 +18,8 @@ config: {'_wandb': {}}
18
  2026-08-12 23:37:05,310 INFO MainThread:3978000 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 100, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'glu', 'activation': 'linear', 'waleed_beta': 10.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/glu-linear-100L_run', 'per_device_train_batch_size': 64, 'num_train_epochs': 1, 'max_steps': 2500, 'learning_rate': 0.0007, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 500, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-glu-linear-100L-16.9M-20260812-233703', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 2498, 'eval_delay': 0, 'per_device_eval_batch_size': 1024, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-glu-linear-100L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
19
  2026-08-12 23:37:05,315 INFO MainThread:3978000 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 16934016 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x14d9ca094510>>
20
  2026-08-12 23:37:05,315 INFO MainThread:3978000 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 16934016 None
 
 
 
 
 
 
18
  2026-08-12 23:37:05,310 INFO MainThread:3978000 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 100, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'glu', 'activation': 'linear', 'waleed_beta': 10.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/glu-linear-100L_run', 'per_device_train_batch_size': 64, 'num_train_epochs': 1, 'max_steps': 2500, 'learning_rate': 0.0007, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 500, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-glu-linear-100L-16.9M-20260812-233703', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 2498, 'eval_delay': 0, 'per_device_eval_batch_size': 1024, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-glu-linear-100L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
19
  2026-08-12 23:37:05,315 INFO MainThread:3978000 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 16934016 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x14d9ca094510>>
20
  2026-08-12 23:37:05,315 INFO MainThread:3978000 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 16934016 None
21
+ 2026-08-12 23:49:56,215 INFO MainThread:3978000 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/research3/tpo5j00e
22
+ 2026-08-12 23:49:56,216 INFO MainThread:3978000 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0
23
+ 2026-08-12 23:49:56,216 INFO MainThread:3978000 [wandb_run.py:_restore():2570] restore
24
+ 2026-08-12 23:49:56,216 INFO MainThread:3978000 [wandb_run.py:_restore():2576] restore done
25
+ 2026-08-12 23:49:56,704 INFO MainThread:3978000 [wandb_run.py:_footer_sync_info():3993] logging synced files
zain/Activation/wandb/run-20260812_233704-tpo5j00e/run-tpo5j00e.wandb CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0af7a2a6ab4c2f8e74e2aaf695ba41a618c3432bd26271f4dbfd0dce107cc40f
3
- size 655360
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0e8c372abca862a576b74703502d7d4917ae126c385be1268175a7ac663bf48a
3
+ size 877846