w-ahmad commited on
Commit
ff5f840
Β·
verified Β·
1 Parent(s): 4051384

Auto upload zain 2026-08-14T21:35:16.328459 (part 4)

Browse files
zain/Activation/out/sweep_summary.json CHANGED
@@ -1,51 +1,37 @@
1
  [
2
  {
3
- "variant": "glu-waleed10-9L",
4
- "eval_loss": 2.248227596282959,
5
- "out": "out/glu-waleed10-9L_run",
6
- "run_name": "LM-glu-waleed10-9L-2.0M-20260814-200424",
7
  "status": "success"
8
  },
9
  {
10
- "variant": "glu-silu-waleed10-9L",
11
- "eval_loss": 2.241689443588257,
12
- "out": "out/glu-silu-waleed10-9L_run",
13
- "run_name": "LM-glu-silu-waleed10-9L-2.0M-20260814-201615",
14
  "status": "success"
15
  },
16
  {
17
- "variant": "mlp-waleed10-9L",
18
- "eval_loss": 2.6366770267486572,
19
- "out": "out/mlp-waleed10-9L_run",
20
- "run_name": "LM-mlp-waleed10-9L-2.0M-20260814-202806",
21
  "status": "success"
22
  },
23
  {
24
- "variant": "mlp-silu-waleed10-9L",
25
- "eval_loss": 2.4668548107147217,
26
- "out": "out/mlp-silu-waleed10-9L_run",
27
- "run_name": "LM-mlp-silu-waleed10-9L-2.0M-20260814-203944",
28
  "status": "success"
29
  },
30
  {
31
- "variant": "glu-waleed-9L",
32
- "eval_loss": 2.2454068660736084,
33
- "out": "out/glu-waleed-9L_run",
34
- "run_name": "LM-glu-waleed-9L-2.0M-20260814-205134",
35
- "status": "success"
36
- },
37
- {
38
- "variant": "glu-situglu_low-9L",
39
- "eval_loss": 2.262706995010376,
40
- "out": "out/glu-situglu_low-9L_run",
41
- "run_name": "LM-glu-situglu_low-9L-2.0M-20260814-210155",
42
- "status": "success"
43
- },
44
- {
45
- "variant": "glu-waleedglu_low-9L",
46
- "eval_loss": 2.23911452293396,
47
- "out": "out/glu-waleedglu_low-9L_run",
48
- "run_name": "LM-glu-waleedglu_low-9L-2.0M-20260814-211407",
49
  "status": "success"
50
  }
51
  ]
 
1
  [
2
  {
3
+ "variant": "glu-relu-100L",
4
+ "eval_loss": 8.116716384887695,
5
+ "out": "out/glu-relu-100L_trash_run",
6
+ "run_name": "LM-glu-relu-100L-16.9M-20260814-210108",
7
  "status": "success"
8
  },
9
  {
10
+ "variant": "glu-gelu-100L",
11
+ "eval_loss": 8.100676536560059,
12
+ "out": "out/glu-gelu-100L_trash_run",
13
+ "run_name": "LM-glu-gelu-100L-16.9M-20260814-210807",
14
  "status": "success"
15
  },
16
  {
17
+ "variant": "glu-sigmoid-100L",
18
+ "eval_loss": 8.628765106201172,
19
+ "out": "out/glu-sigmoid-100L_trash_run",
20
+ "run_name": "LM-glu-sigmoid-100L-16.9M-20260814-211506",
21
  "status": "success"
22
  },
23
  {
24
+ "variant": "glu-waleed10-100L",
25
+ "eval_loss": 7.94517707824707,
26
+ "out": "out/glu-waleed10-100L_trash_run",
27
+ "run_name": "LM-glu-waleed10-100L-16.9M-20260814-212207",
28
  "status": "success"
29
  },
30
  {
31
+ "variant": "glu-silu-waleed10-100L",
32
+ "eval_loss": 8.10374927520752,
33
+ "out": "out/glu-silu-waleed10-100L_trash_run",
34
+ "run_name": "LM-glu-silu-waleed10-100L-16.9M-20260814-212834",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
35
  "status": "success"
36
  }
37
  ]
zain/Activation/wandb/debug-internal.log CHANGED
@@ -1,47 +1,37 @@
1
- {"time":"2026-08-14T21:28:35.344927311Z","level":"INFO","msg":"wandb-core"}
2
- {"time":"2026-08-14T21:28:35.34506265Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"}
3
- {"time":"2026-08-14T21:28:35.602855263Z","level":"INFO","msg":"stream: created new stream","id":"gqskne0x"}
4
- {"time":"2026-08-14T21:28:35.602933409Z","level":"INFO","msg":"handler: started"}
5
- {"time":"2026-08-14T21:28:35.603016258Z","level":"INFO","msg":"stream: started"}
6
- {"time":"2026-08-14T21:28:35.603043817Z","level":"INFO","msg":"writer: started","stream_id":"gqskne0x"}
7
- {"time":"2026-08-14T21:28:35.60305765Z","level":"INFO","msg":"sender: started"}
8
- {"time":"2026-08-14T21:28:36.090276496Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1}
9
- {"time":"2026-08-14T21:28:36.211652827Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
10
- {"time":"2026-08-14T21:28:51.09126529Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":1,"events_offset":0,"events_lines":1,"console_offset":0,"console_lines":3,"uploaded_len":2}
11
- {"time":"2026-08-14T21:28:51.261717283Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
12
- {"time":"2026-08-14T21:28:53.048390257Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":3104}
13
- {"time":"2026-08-14T21:28:53.048417504Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1}
14
- {"time":"2026-08-14T21:28:53.050441969Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":3665}
15
- {"time":"2026-08-14T21:28:53.050479366Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1}
16
- {"time":"2026-08-14T21:28:53.051308024Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":3981}
17
- {"time":"2026-08-14T21:28:53.052248742Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":269}
18
- {"time":"2026-08-14T21:28:53.052348345Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":4293}
19
- {"time":"2026-08-14T21:28:53.053710973Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":419}
20
- {"time":"2026-08-14T21:28:53.054973484Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":5282}
21
- {"time":"2026-08-14T21:28:53.05654974Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":576}
22
- {"time":"2026-08-14T21:28:53.057424589Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":6166}
23
- {"time":"2026-08-14T21:28:53.058161238Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":225}
24
- {"time":"2026-08-14T21:28:53.059182778Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":6765}
25
- {"time":"2026-08-14T21:28:53.059372384Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":41}
26
- {"time":"2026-08-14T21:28:53.059719473Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":6917}
27
- {"time":"2026-08-14T21:28:53.059850702Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":21}
28
- {"time":"2026-08-14T21:28:53.061202591Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":7342}
29
- {"time":"2026-08-14T21:28:53.061276823Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":12}
30
- {"time":"2026-08-14T21:28:53.068366857Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":9009}
31
- {"time":"2026-08-14T21:28:53.068403516Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1}
32
- {"time":"2026-08-14T21:28:53.070211602Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":9602}
33
- {"time":"2026-08-14T21:28:53.071216814Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":294}
34
- {"time":"2026-08-14T21:28:53.071521294Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":10000}
35
- {"time":"2026-08-14T21:28:53.072106936Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":190}
36
- {"time":"2026-08-14T21:28:53.075408887Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":11406}
37
- {"time":"2026-08-14T21:28:53.076614643Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":333}
38
- {"time":"2026-08-14T21:29:06.110603465Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":1,"history_lines":2,"events_offset":1,"events_lines":2,"console_offset":1,"console_lines":1}
39
- {"time":"2026-08-14T21:29:06.83621196Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
40
- {"time":"2026-08-14T21:29:21.108433332Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":3,"history_lines":3,"events_offset":3,"events_lines":2,"console_offset":1,"console_lines":1}
41
- {"time":"2026-08-14T21:29:21.999015678Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
42
- {"time":"2026-08-14T21:29:36.110784946Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":6,"history_lines":2,"events_offset":5,"events_lines":2,"console_offset":3,"console_lines":12}
43
- {"time":"2026-08-14T21:29:36.946170207Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
44
- {"time":"2026-08-14T21:29:51.109458297Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":8,"history_lines":3,"events_offset":7,"events_lines":2,"console_offset":11,"console_lines":1}
45
- {"time":"2026-08-14T21:29:52.226871571Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
46
- {"time":"2026-08-14T21:30:06.110393089Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":11,"history_lines":2,"events_offset":9,"events_lines":2,"console_offset":15,"console_lines":10}
47
- {"time":"2026-08-14T21:30:07.570174219Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
 
1
+ {"time":"2026-08-14T21:31:53.802248642Z","level":"INFO","msg":"wandb-core"}
2
+ {"time":"2026-08-14T21:31:53.802536305Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"}
3
+ {"time":"2026-08-14T21:31:54.062231678Z","level":"INFO","msg":"stream: created new stream","id":"q9vcmje5"}
4
+ {"time":"2026-08-14T21:31:54.062319994Z","level":"INFO","msg":"handler: started"}
5
+ {"time":"2026-08-14T21:31:54.062390309Z","level":"INFO","msg":"stream: started"}
6
+ {"time":"2026-08-14T21:31:54.062412718Z","level":"INFO","msg":"writer: started","stream_id":"q9vcmje5"}
7
+ {"time":"2026-08-14T21:31:54.062420181Z","level":"INFO","msg":"sender: started"}
8
+ {"time":"2026-08-14T21:31:54.416838412Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1}
9
+ {"time":"2026-08-14T21:31:54.510348438Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
10
+ {"time":"2026-08-14T21:32:09.41771802Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":0,"events_lines":1,"console_offset":0,"console_lines":3,"uploaded_len":2}
11
+ {"time":"2026-08-14T21:32:09.578280298Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
12
+ {"time":"2026-08-14T21:32:24.417605975Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":1,"events_offset":1,"events_lines":2,"console_offset":2,"console_lines":2}
13
+ {"time":"2026-08-14T21:32:24.549935438Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
14
+ {"time":"2026-08-14T21:32:39.41719538Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":1,"history_lines":1,"events_offset":3,"events_lines":2,"console_offset":2,"console_lines":1}
15
+ {"time":"2026-08-14T21:32:39.566802488Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
16
+ {"time":"2026-08-14T21:32:54.417058222Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":5,"events_lines":2,"console_offset":2,"console_lines":1}
17
+ {"time":"2026-08-14T21:32:54.526885785Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
18
+ {"time":"2026-08-14T21:33:09.417361811Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":2,"history_lines":1,"events_offset":7,"events_lines":2,"console_offset":2,"console_lines":1}
19
+ {"time":"2026-08-14T21:33:09.541292122Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
20
+ {"time":"2026-08-14T21:33:24.41766197Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":3,"history_lines":1,"events_offset":9,"events_lines":2,"console_offset":2,"console_lines":1}
21
+ {"time":"2026-08-14T21:33:24.549198743Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
22
+ {"time":"2026-08-14T21:33:39.417110044Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":11,"events_lines":2,"console_offset":2,"console_lines":1}
23
+ {"time":"2026-08-14T21:33:39.551776959Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
24
+ {"time":"2026-08-14T21:33:54.417538658Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":4,"history_lines":1,"events_offset":13,"events_lines":2,"console_offset":2,"console_lines":1}
25
+ {"time":"2026-08-14T21:33:54.555356936Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
26
+ {"time":"2026-08-14T21:34:09.417890895Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":5,"history_lines":1,"events_offset":15,"events_lines":2,"console_offset":4,"console_lines":10}
27
+ {"time":"2026-08-14T21:34:09.592684206Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
28
+ {"time":"2026-08-14T21:34:24.417179414Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":6,"history_lines":1,"events_offset":17,"events_lines":2,"console_offset":12,"console_lines":1}
29
+ {"time":"2026-08-14T21:34:24.534710617Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
30
+ {"time":"2026-08-14T21:34:39.417073445Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":19,"events_lines":2,"console_offset":12,"console_lines":1}
31
+ {"time":"2026-08-14T21:34:39.570434803Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
32
+ {"time":"2026-08-14T21:34:54.417818476Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":7,"history_lines":1,"events_offset":21,"events_lines":2,"console_offset":12,"console_lines":1}
33
+ {"time":"2026-08-14T21:34:54.541208219Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
34
+ {"time":"2026-08-14T21:35:09.41711858Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":8,"history_lines":1,"events_offset":23,"events_lines":2,"console_offset":12,"console_lines":1}
35
+ {"time":"2026-08-14T21:35:09.54503893Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
36
+ {"time":"2026-08-14T21:35:24.417709905Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":9,"history_lines":1,"events_offset":25,"events_lines":2,"console_offset":12,"console_lines":1}
37
+ {"time":"2026-08-14T21:35:24.538800467Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
 
 
 
 
 
 
 
 
 
 
zain/Activation/wandb/debug.log CHANGED
@@ -1,20 +1,22 @@
1
- 2026-08-14 21:28:35,343 INFO MainThread:936813 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260814_212835-gqskne0x/logs/debug.log
2
- 2026-08-14 21:28:35,343 INFO MainThread:936813 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260814_212835-gqskne0x/logs/debug-internal.log
3
- 2026-08-14 21:28:35,343 INFO MainThread:936813 [wandb_init.py:init():772] calling init triggers
4
- 2026-08-14 21:28:35,343 INFO MainThread:936813 [wandb_init.py:init():777] wandb.init called with sweep_config: {}
 
 
 
5
  config: {'_wandb': {}}
6
- 2026-08-14 21:28:35,343 INFO MainThread:936813 [wandb_init.py:init():820] starting backend
7
- 2026-08-14 21:28:35,343 INFO MainThread:936813 [wandb_init.py:init():826] Connected to an existing wandb-core service via WANDB_SERVICE
8
- 2026-08-14 21:28:35,343 INFO MainThread:936813 [wandb_init.py:init():835] sending inform_init request
9
- 2026-08-14 21:28:35,603 INFO MainThread:936813 [wandb_init.py:init():840] backend started and connected
10
- 2026-08-14 21:28:35,605 INFO MainThread:936813 [wandb_init.py:init():910] updated telemetry
11
- 2026-08-14 21:28:35,614 INFO MainThread:936813 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout
12
- 2026-08-14 21:28:35,996 INFO MainThread:936813 [wandb_init.py:init():978] starting run threads in backend
13
- 2026-08-14 21:28:36,075 INFO MainThread:936813 [wandb_run.py:_console_start():2621] atexit reg
14
- 2026-08-14 21:28:36,075 INFO MainThread:936813 [wandb_run.py:_redirect():2471] redirect: wrap_raw
15
- 2026-08-14 21:28:36,076 INFO MainThread:936813 [wandb_run.py:_redirect():2540] Wrapping output streams.
16
- 2026-08-14 21:28:36,076 INFO MainThread:936813 [wandb_run.py:_redirect():2563] Redirects installed.
17
- 2026-08-14 21:28:36,077 INFO MainThread:936813 [wandb_init.py:init():1016] run started, returning control to user process
18
- 2026-08-14 21:28:36,078 INFO MainThread:936813 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 100, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'glu', 'activation': 'silu-waleed10', 'waleed_beta': 4, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/glu-silu-waleed10-100L_trash_run', 'per_device_train_batch_size': 64, 'num_train_epochs': 1, 'max_steps': 3500, 'learning_rate': 9e-05, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 50, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-glu-silu-waleed10-100L-16.9M-20260814-212834', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 2498, 'eval_delay': 0, 'per_device_eval_batch_size': 1024, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 50, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-glu-silu-waleed10-100L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
19
- 2026-08-14 21:28:36,083 INFO MainThread:936813 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 16934016 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x14ee399dda90>>
20
- 2026-08-14 21:28:36,083 INFO MainThread:936813 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 16934016 None
 
1
+ 2026-08-14 21:31:53,800 INFO MainThread:1229423 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1
2
+ 2026-08-14 21:31:53,800 INFO MainThread:1229423 [wandb_setup.py:_flush():81] Configure stats pid to 1229423
3
+ 2026-08-14 21:31:53,800 INFO MainThread:1229423 [wandb_setup.py:_flush():81] Loading settings from environment variables
4
+ 2026-08-14 21:31:53,800 INFO MainThread:1229423 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260814_213153-q9vcmje5/logs/debug.log
5
+ 2026-08-14 21:31:53,800 INFO MainThread:1229423 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260814_213153-q9vcmje5/logs/debug-internal.log
6
+ 2026-08-14 21:31:53,800 INFO MainThread:1229423 [wandb_init.py:init():772] calling init triggers
7
+ 2026-08-14 21:31:53,801 INFO MainThread:1229423 [wandb_init.py:init():777] wandb.init called with sweep_config: {}
8
  config: {'_wandb': {}}
9
+ 2026-08-14 21:31:53,801 INFO MainThread:1229423 [wandb_init.py:init():820] starting backend
10
+ 2026-08-14 21:31:53,801 INFO MainThread:1229423 [wandb_init.py:init():835] sending inform_init request
11
+ 2026-08-14 21:31:54,062 INFO MainThread:1229423 [wandb_init.py:init():840] backend started and connected
12
+ 2026-08-14 21:31:54,063 INFO MainThread:1229423 [wandb_init.py:init():910] updated telemetry
13
+ 2026-08-14 21:31:54,070 INFO MainThread:1229423 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout
14
+ 2026-08-14 21:31:54,336 INFO MainThread:1229423 [wandb_init.py:init():978] starting run threads in backend
15
+ 2026-08-14 21:31:54,409 INFO MainThread:1229423 [wandb_run.py:_console_start():2621] atexit reg
16
+ 2026-08-14 21:31:54,409 INFO MainThread:1229423 [wandb_run.py:_redirect():2471] redirect: wrap_raw
17
+ 2026-08-14 21:31:54,409 INFO MainThread:1229423 [wandb_run.py:_redirect():2540] Wrapping output streams.
18
+ 2026-08-14 21:31:54,409 INFO MainThread:1229423 [wandb_run.py:_redirect():2563] Redirects installed.
19
+ 2026-08-14 21:31:54,412 INFO MainThread:1229423 [wandb_init.py:init():1016] run started, returning control to user process
20
+ 2026-08-14 21:31:54,413 INFO MainThread:1229423 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 21, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'glu', 'activation': 'waleed10', 'waleed_beta': 10.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/glu-waleed10-21L_run', 'per_device_train_batch_size': 128, 'num_train_epochs': 1, 'max_steps': 1000, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant', 'lr_scheduler_kwargs': None, 'warmup_steps': 0, 'optim': 'adamw_torch', 'optim_args': None, 'weight_decay': 0.01, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 4, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-glu-waleed10-21L-4.0M-20260814-213152', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 60000, 'eval_delay': 0, 'per_device_eval_batch_size': 128, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/A-glu-waleed10-21L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
21
+ 2026-08-14 21:31:54,415 INFO MainThread:1229423 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 3970432 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x1528c060ecd0>>
22
+ 2026-08-14 21:31:54,415 INFO MainThread:1229423 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 3970432 None
 
zain/Activation/wandb/run-20260814_193907-kk1xbqht/logs/debug-core.log CHANGED
@@ -75,3 +75,11 @@
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
 
 
 
 
 
 
 
 
 
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
78
+ {"time":"2026-08-14T21:35:05.007052341Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
79
+ {"time":"2026-08-14T21:35:06.745707472Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
80
+ {"time":"2026-08-14T21:35:06.767299957Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"gqskne0x","id":"3(@)"}
81
+ {"time":"2026-08-14T21:35:06.773662133Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"gqskne0x","id":"3(@)"}
82
+ {"time":"2026-08-14T21:35:08.806158062Z","level":"INFO","msg":"connection: closing","id":"3(@)"}
83
+ {"time":"2026-08-14T21:35:08.806241813Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"}
84
+ {"time":"2026-08-14T21:35:08.806161779Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"}
85
+ {"time":"2026-08-14T21:35:08.806251477Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"}
zain/Activation/wandb/run-20260814_195151-ix6wcbwv/logs/debug-core.log CHANGED
@@ -75,3 +75,11 @@
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
 
 
 
 
 
 
 
 
 
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
78
+ {"time":"2026-08-14T21:35:05.007052341Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
79
+ {"time":"2026-08-14T21:35:06.745707472Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
80
+ {"time":"2026-08-14T21:35:06.767299957Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"gqskne0x","id":"3(@)"}
81
+ {"time":"2026-08-14T21:35:06.773662133Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"gqskne0x","id":"3(@)"}
82
+ {"time":"2026-08-14T21:35:08.806158062Z","level":"INFO","msg":"connection: closing","id":"3(@)"}
83
+ {"time":"2026-08-14T21:35:08.806241813Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"}
84
+ {"time":"2026-08-14T21:35:08.806161779Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"}
85
+ {"time":"2026-08-14T21:35:08.806251477Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"}
zain/Activation/wandb/run-20260814_200503-bdhnr27j/logs/debug-core.log CHANGED
@@ -75,3 +75,11 @@
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
 
 
 
 
 
 
 
 
 
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
78
+ {"time":"2026-08-14T21:35:05.007052341Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
79
+ {"time":"2026-08-14T21:35:06.745707472Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
80
+ {"time":"2026-08-14T21:35:06.767299957Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"gqskne0x","id":"3(@)"}
81
+ {"time":"2026-08-14T21:35:06.773662133Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"gqskne0x","id":"3(@)"}
82
+ {"time":"2026-08-14T21:35:08.806158062Z","level":"INFO","msg":"connection: closing","id":"3(@)"}
83
+ {"time":"2026-08-14T21:35:08.806241813Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"}
84
+ {"time":"2026-08-14T21:35:08.806161779Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"}
85
+ {"time":"2026-08-14T21:35:08.806251477Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"}
zain/Activation/wandb/run-20260814_202128-twgo1e89/logs/debug-core.log CHANGED
@@ -75,3 +75,11 @@
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
 
 
 
 
 
 
 
 
 
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
78
+ {"time":"2026-08-14T21:35:05.007052341Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
79
+ {"time":"2026-08-14T21:35:06.745707472Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
80
+ {"time":"2026-08-14T21:35:06.767299957Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"gqskne0x","id":"3(@)"}
81
+ {"time":"2026-08-14T21:35:06.773662133Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"gqskne0x","id":"3(@)"}
82
+ {"time":"2026-08-14T21:35:08.806158062Z","level":"INFO","msg":"connection: closing","id":"3(@)"}
83
+ {"time":"2026-08-14T21:35:08.806241813Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"}
84
+ {"time":"2026-08-14T21:35:08.806161779Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"}
85
+ {"time":"2026-08-14T21:35:08.806251477Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"}
zain/Activation/wandb/run-20260814_203803-7p5140bq/logs/debug-core.log CHANGED
@@ -75,3 +75,11 @@
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
 
 
 
 
 
 
 
 
 
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
78
+ {"time":"2026-08-14T21:35:05.007052341Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
79
+ {"time":"2026-08-14T21:35:06.745707472Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
80
+ {"time":"2026-08-14T21:35:06.767299957Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"gqskne0x","id":"3(@)"}
81
+ {"time":"2026-08-14T21:35:06.773662133Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"gqskne0x","id":"3(@)"}
82
+ {"time":"2026-08-14T21:35:08.806158062Z","level":"INFO","msg":"connection: closing","id":"3(@)"}
83
+ {"time":"2026-08-14T21:35:08.806241813Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"}
84
+ {"time":"2026-08-14T21:35:08.806161779Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"}
85
+ {"time":"2026-08-14T21:35:08.806251477Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"}
zain/Activation/wandb/run-20260814_210109-qauftiuh/logs/debug-core.log CHANGED
@@ -75,3 +75,11 @@
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
 
 
 
 
 
 
 
 
 
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
78
+ {"time":"2026-08-14T21:35:05.007052341Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
79
+ {"time":"2026-08-14T21:35:06.745707472Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
80
+ {"time":"2026-08-14T21:35:06.767299957Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"gqskne0x","id":"3(@)"}
81
+ {"time":"2026-08-14T21:35:06.773662133Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"gqskne0x","id":"3(@)"}
82
+ {"time":"2026-08-14T21:35:08.806158062Z","level":"INFO","msg":"connection: closing","id":"3(@)"}
83
+ {"time":"2026-08-14T21:35:08.806241813Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"}
84
+ {"time":"2026-08-14T21:35:08.806161779Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"}
85
+ {"time":"2026-08-14T21:35:08.806251477Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"}
zain/Activation/wandb/run-20260814_210809-a6574zx8/logs/debug-core.log CHANGED
@@ -75,3 +75,11 @@
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
 
 
 
 
 
 
 
 
 
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
78
+ {"time":"2026-08-14T21:35:05.007052341Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
79
+ {"time":"2026-08-14T21:35:06.745707472Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
80
+ {"time":"2026-08-14T21:35:06.767299957Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"gqskne0x","id":"3(@)"}
81
+ {"time":"2026-08-14T21:35:06.773662133Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"gqskne0x","id":"3(@)"}
82
+ {"time":"2026-08-14T21:35:08.806158062Z","level":"INFO","msg":"connection: closing","id":"3(@)"}
83
+ {"time":"2026-08-14T21:35:08.806241813Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"}
84
+ {"time":"2026-08-14T21:35:08.806161779Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"}
85
+ {"time":"2026-08-14T21:35:08.806251477Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"}
zain/Activation/wandb/run-20260814_211507-54wnyexz/logs/debug-core.log CHANGED
@@ -75,3 +75,11 @@
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
 
 
 
 
 
 
 
 
 
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
78
+ {"time":"2026-08-14T21:35:05.007052341Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
79
+ {"time":"2026-08-14T21:35:06.745707472Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
80
+ {"time":"2026-08-14T21:35:06.767299957Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"gqskne0x","id":"3(@)"}
81
+ {"time":"2026-08-14T21:35:06.773662133Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"gqskne0x","id":"3(@)"}
82
+ {"time":"2026-08-14T21:35:08.806158062Z","level":"INFO","msg":"connection: closing","id":"3(@)"}
83
+ {"time":"2026-08-14T21:35:08.806241813Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"}
84
+ {"time":"2026-08-14T21:35:08.806161779Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"}
85
+ {"time":"2026-08-14T21:35:08.806251477Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"}
zain/Activation/wandb/run-20260814_212208-n6fjpg9o/logs/debug-core.log CHANGED
@@ -75,3 +75,11 @@
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
 
 
 
 
 
 
 
 
 
75
  {"time":"2026-08-14T21:28:35.344777056Z","level":"INFO","msg":"handleInformInit: received","streamId":"gqskne0x","id":"3(@)"}
76
  {"time":"2026-08-14T21:28:35.603024157Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"gqskne0x","id":"3(@)"}
77
  {"time":"2026-08-14T21:28:41.078893683Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
78
+ {"time":"2026-08-14T21:35:05.007052341Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
79
+ {"time":"2026-08-14T21:35:06.745707472Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"yxbo55z35lse"}
80
+ {"time":"2026-08-14T21:35:06.767299957Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"gqskne0x","id":"3(@)"}
81
+ {"time":"2026-08-14T21:35:06.773662133Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"gqskne0x","id":"3(@)"}
82
+ {"time":"2026-08-14T21:35:08.806158062Z","level":"INFO","msg":"connection: closing","id":"3(@)"}
83
+ {"time":"2026-08-14T21:35:08.806241813Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"}
84
+ {"time":"2026-08-14T21:35:08.806161779Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"}
85
+ {"time":"2026-08-14T21:35:08.806251477Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"}
zain/Activation/wandb/run-20260814_212835-gqskne0x/files/config.yaml ADDED
@@ -0,0 +1,439 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _name_or_path:
2
+ value: ""
3
+ _wandb:
4
+ value:
5
+ cli_version: 0.28.1
6
+ e:
7
+ pf2h3cdzqvkrfjeqx7v0rn4hp1f52wde:
8
+ args:
9
+ - --config
10
+ - /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/configs/baseline100l.yaml
11
+ - --variants
12
+ - glu-relu
13
+ - glu-gelu
14
+ - glu-sigmoid
15
+ - glu-waleed10
16
+ - glu-silu-waleed10
17
+ codePath: sweep.py
18
+ codePathLocal: sweep.py
19
+ cpu_count: 112
20
+ cpu_count_logical: 224
21
+ cudaVersion: "12.4"
22
+ disk:
23
+ /:
24
+ total: "1560765693952"
25
+ used: "709055246336"
26
+ email: deepnevro@gmail.com
27
+ executable: /mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python
28
+ git:
29
+ commit: 3fb47c6943d6927eafc39bd399a3d3a520bae74a
30
+ remote: https://github.com/w-ahmad1a10/Activation.git
31
+ gpu: NVIDIA H100 80GB HBM3
32
+ gpu_count: 8
33
+ gpu_nvidia:
34
+ - architecture: Hopper
35
+ cudaCores: 16896
36
+ memoryTotal: "85520809984"
37
+ name: NVIDIA H100 80GB HBM3
38
+ uuid: GPU-39c684a5-fde6-83d7-1663-0859795881ae
39
+ - architecture: Hopper
40
+ cudaCores: 16896
41
+ memoryTotal: "85520809984"
42
+ name: NVIDIA H100 80GB HBM3
43
+ uuid: GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3
44
+ - architecture: Hopper
45
+ cudaCores: 16896
46
+ memoryTotal: "85520809984"
47
+ name: NVIDIA H100 80GB HBM3
48
+ uuid: GPU-132944c4-b689-2b5f-89a4-d730401677ab
49
+ - architecture: Hopper
50
+ cudaCores: 16896
51
+ memoryTotal: "85520809984"
52
+ name: NVIDIA H100 80GB HBM3
53
+ uuid: GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864
54
+ - architecture: Hopper
55
+ cudaCores: 16896
56
+ memoryTotal: "85520809984"
57
+ name: NVIDIA H100 80GB HBM3
58
+ uuid: GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef
59
+ - architecture: Hopper
60
+ cudaCores: 16896
61
+ memoryTotal: "85520809984"
62
+ name: NVIDIA H100 80GB HBM3
63
+ uuid: GPU-bc6c3e3c-9b90-09ca-c034-774961847c54
64
+ - architecture: Hopper
65
+ cudaCores: 16896
66
+ memoryTotal: "85520809984"
67
+ name: NVIDIA H100 80GB HBM3
68
+ uuid: GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9
69
+ - architecture: Hopper
70
+ cudaCores: 16896
71
+ memoryTotal: "85520809984"
72
+ name: NVIDIA H100 80GB HBM3
73
+ uuid: GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea
74
+ host: deeplens-k3s-node1
75
+ memory:
76
+ total: "2164089937920"
77
+ os: Linux-5.15.0-126-generic-x86_64-with-glibc2.35
78
+ program: /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/sweep.py
79
+ python: CPython 3.11.15
80
+ root: /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation
81
+ startedAt: "2026-08-14T21:28:35.342440Z"
82
+ writerId: pf2h3cdzqvkrfjeqx7v0rn4hp1f52wde
83
+ m:
84
+ - "1": train/global_step
85
+ "6":
86
+ - 3
87
+ "7": []
88
+ - "2": '*'
89
+ "5": 1
90
+ "6":
91
+ - 1
92
+ "7": []
93
+ python_version: 3.11.15
94
+ t:
95
+ "1":
96
+ - 1
97
+ - 5
98
+ - 11
99
+ - 41
100
+ - 49
101
+ - 51
102
+ - 53
103
+ - 71
104
+ "2":
105
+ - 1
106
+ - 5
107
+ - 11
108
+ - 41
109
+ - 49
110
+ - 51
111
+ - 53
112
+ - 71
113
+ "3":
114
+ - 2
115
+ - 7
116
+ - 13
117
+ - 19
118
+ - 41
119
+ - 61
120
+ - 62
121
+ - 66
122
+ "4": 3.11.15
123
+ "5": 0.28.1
124
+ "6": 5.16.0.dev0
125
+ "9":
126
+ "1": transformers_trainer
127
+ "12": 0.28.1
128
+ "13": linux-x86_64
129
+ accelerator_config:
130
+ value:
131
+ dispatch_batches: null
132
+ even_batches: true
133
+ gradient_accumulation_kwargs: null
134
+ non_blocking: false
135
+ split_batches: false
136
+ use_seedable_sampler: true
137
+ activation:
138
+ value: silu-waleed10
139
+ adam_beta1:
140
+ value: 0.9
141
+ adam_beta2:
142
+ value: 0.999
143
+ adam_epsilon:
144
+ value: 1e-08
145
+ architectures:
146
+ value: null
147
+ attention_bias:
148
+ value: false
149
+ attention_dropout:
150
+ value: 0
151
+ auto_find_batch_size:
152
+ value: false
153
+ average_tokens_across_devices:
154
+ value: true
155
+ batch_eval_metrics:
156
+ value: false
157
+ bf16:
158
+ value: true
159
+ bf16_full_eval:
160
+ value: false
161
+ bos_token_id:
162
+ value: 1
163
+ chunk_size_feed_forward:
164
+ value: 0
165
+ data_seed:
166
+ value: 42
167
+ dataloader_drop_last:
168
+ value: false
169
+ dataloader_in_order:
170
+ value: true
171
+ dataloader_multiprocessing_context:
172
+ value: null
173
+ dataloader_num_workers:
174
+ value: 0
175
+ dataloader_persistent_workers:
176
+ value: false
177
+ dataloader_pin_memory:
178
+ value: true
179
+ dataloader_prefetch_factor:
180
+ value: null
181
+ ddp_backend:
182
+ value: null
183
+ ddp_broadcast_buffers:
184
+ value: null
185
+ ddp_bucket_cap_mb:
186
+ value: null
187
+ ddp_find_unused_parameters:
188
+ value: null
189
+ ddp_static_graph:
190
+ value: null
191
+ ddp_timeout:
192
+ value: 1800
193
+ debug:
194
+ value: []
195
+ deepspeed:
196
+ value: null
197
+ disable_tqdm:
198
+ value: false
199
+ do_eval:
200
+ value: true
201
+ do_predict:
202
+ value: false
203
+ do_train:
204
+ value: false
205
+ dtype:
206
+ value: null
207
+ enable_jit_checkpoint:
208
+ value: false
209
+ eos_token_id:
210
+ value: 2
211
+ eval_accumulation_steps:
212
+ value: null
213
+ eval_delay:
214
+ value: 0
215
+ eval_do_concat_batches:
216
+ value: true
217
+ eval_on_start:
218
+ value: false
219
+ eval_steps:
220
+ value: 2498
221
+ eval_strategy:
222
+ value: steps
223
+ eval_use_gather_object:
224
+ value: false
225
+ fp16:
226
+ value: false
227
+ fp16_full_eval:
228
+ value: false
229
+ fsdp:
230
+ value: null
231
+ fsdp_config:
232
+ value: null
233
+ full_determinism:
234
+ value: false
235
+ gradient_accumulation_steps:
236
+ value: 1
237
+ gradient_checkpointing:
238
+ value: false
239
+ gradient_checkpointing_kwargs:
240
+ value: null
241
+ greater_is_better:
242
+ value: null
243
+ head_dim:
244
+ value: 32
245
+ hidden_act:
246
+ value: silu
247
+ hidden_size:
248
+ value: 128
249
+ hub_always_push:
250
+ value: false
251
+ hub_model_id:
252
+ value: w-ahmad/6L-glu-silu-waleed10-100L
253
+ hub_private_repo:
254
+ value: null
255
+ hub_revision:
256
+ value: null
257
+ hub_strategy:
258
+ value: every_save
259
+ hub_token:
260
+ value: <HUB_TOKEN>
261
+ id2label:
262
+ value:
263
+ "0": LABEL_0
264
+ "1": LABEL_1
265
+ ignore_data_skip:
266
+ value: false
267
+ include_for_metrics:
268
+ value: []
269
+ include_num_input_tokens_seen:
270
+ value: "no"
271
+ initializer_range:
272
+ value: 0.02
273
+ intermediate_size:
274
+ value: 256
275
+ is_encoder_decoder:
276
+ value: false
277
+ label_names:
278
+ value: null
279
+ label_smoothing_factor:
280
+ value: 0
281
+ label2id:
282
+ value:
283
+ LABEL_0: 0
284
+ LABEL_1: 1
285
+ learning_rate:
286
+ value: 9e-05
287
+ length_column_name:
288
+ value: length
289
+ liger_kernel_config:
290
+ value: null
291
+ load_best_model_at_end:
292
+ value: false
293
+ local_rank:
294
+ value: -1
295
+ log_level:
296
+ value: passive
297
+ log_level_replica:
298
+ value: warning
299
+ log_on_each_node:
300
+ value: true
301
+ logging_first_step:
302
+ value: false
303
+ logging_nan_inf_filter:
304
+ value: true
305
+ logging_steps:
306
+ value: 20
307
+ logging_strategy:
308
+ value: steps
309
+ lr_scheduler_kwargs:
310
+ value: null
311
+ lr_scheduler_type:
312
+ value: constant_with_warmup
313
+ max_grad_norm:
314
+ value: 1
315
+ max_position_embeddings:
316
+ value: 512
317
+ max_steps:
318
+ value: 3500
319
+ metric_for_best_model:
320
+ value: null
321
+ mlp_bias:
322
+ value: false
323
+ mlp_type:
324
+ value: glu
325
+ model/num_parameters:
326
+ value: 16934016
327
+ model_type:
328
+ value: tiny_llama
329
+ neftune_noise_alpha:
330
+ value: null
331
+ num_attention_heads:
332
+ value: 4
333
+ num_hidden_layers:
334
+ value: 100
335
+ num_key_value_heads:
336
+ value: 4
337
+ num_train_epochs:
338
+ value: 1
339
+ optim:
340
+ value: adamw_torch_fused
341
+ optim_args:
342
+ value: null
343
+ optim_target_modules:
344
+ value: null
345
+ output_attentions:
346
+ value: false
347
+ output_dir:
348
+ value: out/glu-silu-waleed10-100L_trash_run
349
+ output_hidden_states:
350
+ value: false
351
+ pad_token_id:
352
+ value: 0
353
+ parallelism_config:
354
+ value: null
355
+ per_device_eval_batch_size:
356
+ value: 1024
357
+ per_device_train_batch_size:
358
+ value: 64
359
+ prediction_loss_only:
360
+ value: false
361
+ pretraining_tp:
362
+ value: 1
363
+ problem_type:
364
+ value: null
365
+ project:
366
+ value: huggingface
367
+ push_to_hub:
368
+ value: false
369
+ remove_unused_columns:
370
+ value: false
371
+ report_to:
372
+ value:
373
+ - wandb
374
+ restore_callback_states_from_checkpoint:
375
+ value: false
376
+ resume_from_checkpoint:
377
+ value: null
378
+ return_dict:
379
+ value: true
380
+ rms_norm_eps:
381
+ value: 1e-06
382
+ rope_parameters:
383
+ value:
384
+ rope_theta: 10000
385
+ rope_type: default
386
+ run_name:
387
+ value: LM-glu-silu-waleed10-100L-16.9M-20260814-212834
388
+ save_on_each_node:
389
+ value: false
390
+ save_only_model:
391
+ value: false
392
+ save_steps:
393
+ value: 50
394
+ save_strategy:
395
+ value: steps
396
+ save_total_limit:
397
+ value: null
398
+ seed:
399
+ value: 42
400
+ skip_memory_metrics:
401
+ value: true
402
+ tf32:
403
+ value: null
404
+ tie_word_embeddings:
405
+ value: true
406
+ tokenizer_name:
407
+ value: w-ahmad/tiny-stories-tokenizer
408
+ torch_compile:
409
+ value: false
410
+ torch_compile_backend:
411
+ value: null
412
+ torch_compile_mode:
413
+ value: null
414
+ torch_empty_cache_steps:
415
+ value: null
416
+ trackio_bucket_id:
417
+ value: null
418
+ trackio_space_id:
419
+ value: null
420
+ trackio_static_space_id:
421
+ value: null
422
+ train_sampling_strategy:
423
+ value: random
424
+ transformers_version:
425
+ value: 5.16.0.dev0
426
+ use_cache:
427
+ value: false
428
+ use_cpu:
429
+ value: false
430
+ use_liger_kernel:
431
+ value: false
432
+ vocab_size:
433
+ value: 4096
434
+ waleed_beta:
435
+ value: 4
436
+ warmup_steps:
437
+ value: 50
438
+ weight_decay:
439
+ value: 0
zain/Activation/wandb/run-20260814_212835-gqskne0x/files/output.log CHANGED
@@ -19,8 +19,93 @@ Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:00<00:00, 15.
19
  - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
20
  - If you are not the owner of the model architecture class, please contact the model code owner to update it.
21
  Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:00<00:00, 13.82it/s]
22
- 80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 2797/3500 [01:34<03:16, 3.57it/s], ?it/s]
23
  {'loss': '8.051', 'grad_norm': '3.817e+08', 'learning_rate': '9e-05', 'epoch': '0.1834', 'train/total_time_seconds': '60.56', 'train/time_per_step_avg': '0.2375', 'train/epoch_time_elapsed': '72.75', 'train/estimated_remaining_minutes': '0.2894', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '185.9', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0009193', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.1', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '-1.162e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3555', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.3555', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3047', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6602', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '1488', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.0188', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '0.5117', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '3.141', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-3.141', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '2.406', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '5.547', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '1454', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.006104', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '0.5039', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '2.469', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-2.469', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '2.469', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '4.938', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '492', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '0.125', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '0.2051', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '1.398', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '-0.7461', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '1.398', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '2.145', 'train/tensor_act_model_layers_0_mlp/norm': '487.8', 'train/tensor_act_model_layers_0_mlp/mean': '0.1245', 'train/tensor_act_model_layers_0_mlp/std': '0.2031', 'train/tensor_act_model_layers_0_mlp/max_abs': '1.344', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '-0.7383', 'train/tensor_act_model_layers_0_mlp/max': '1.344', 'train/tensor_act_model_layers_0_mlp/range': '2.082', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '542.1', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '0.1245', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.2334', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '1.406', 'train/tenso
24
  {'loss': '8.067', 'grad_norm': '3.02e+08', 'learning_rate': '9e-05', 'epoch': '0.1847', 'train/total_time_seconds': '65.12', 'train/time_per_step_avg': '0.2378', 'train/epoch_time_elapsed': '78.35', 'train/estimated_remaining_minutes': '0.301'}
25
  {'loss': '8.077', 'grad_norm': '2.317e+08', 'learning_rate': '9e-05', 'epoch': '0.186', 'train/total_time_seconds': '70.12', 'train/time_per_step_avg': '0.2378', 'train/epoch_time_elapsed': '84.44', 'train/estimated_remaining_minutes': '0.3133', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '186.1', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0008926', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.2', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '-3.07e-06', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3262', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.3262', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.2949', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6211', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '1897', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.03638', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '0.6523', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '4.438', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-4.438', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '3.609', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '8.047', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '1877', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.01044', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '0.6484', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '3.797', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-3.797', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '3.422', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '7.219', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '2102', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '0.6133', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '0.8242', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '4.688', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '-0.8867', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '4.688', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '5.574', 'train/tensor_act_model_layers_0_mlp/norm': '1878', 'train/tensor_act_model_layers_0_mlp/mean': '0.5703', 'train/tensor_act_model_layers_0_mlp/std': '0.7188', 'train/tensor_act_model_layers_0_mlp/max_abs': '3.297', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '-0.8711', 'train/tensor_act_model_layers_0_mlp/max': '3.297', 'train/tensor_act_model_layers_0_mlp/range': '4.168', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '1897', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '0.5703', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.7305', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '3.422', 'train/tensor_
26
  {'loss': '8.085', 'grad_norm': '1.646e+08', 'learning_rate': '9e-05', 'epoch': '0.1874', 'train/total_time_seconds': '74.72', 'train/time_per_step_avg': '0.2379', 'train/epoch_time_elapsed': '90.08', 'train/estimated_remaining_minutes': '0.3225'}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
19
  - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
20
  - If you are not the owner of the model architecture class, please contact the model code owner to update it.
21
  Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:00<00:00, 13.82it/s]
22
+ 80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 2800/3500 [01:35<03:15, 3.59it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From πŸ‘‰v4.50πŸ‘ˆ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
23
  {'loss': '8.051', 'grad_norm': '3.817e+08', 'learning_rate': '9e-05', 'epoch': '0.1834', 'train/total_time_seconds': '60.56', 'train/time_per_step_avg': '0.2375', 'train/epoch_time_elapsed': '72.75', 'train/estimated_remaining_minutes': '0.2894', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '185.9', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0009193', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.1', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '-1.162e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3555', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.3555', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3047', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6602', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '1488', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.0188', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '0.5117', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '3.141', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-3.141', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '2.406', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '5.547', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '1454', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.006104', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '0.5039', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '2.469', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-2.469', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '2.469', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '4.938', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '492', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '0.125', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '0.2051', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '1.398', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '-0.7461', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '1.398', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '2.145', 'train/tensor_act_model_layers_0_mlp/norm': '487.8', 'train/tensor_act_model_layers_0_mlp/mean': '0.1245', 'train/tensor_act_model_layers_0_mlp/std': '0.2031', 'train/tensor_act_model_layers_0_mlp/max_abs': '1.344', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '-0.7383', 'train/tensor_act_model_layers_0_mlp/max': '1.344', 'train/tensor_act_model_layers_0_mlp/range': '2.082', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '542.1', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '0.1245', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.2334', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '1.406', 'train/tenso
24
  {'loss': '8.067', 'grad_norm': '3.02e+08', 'learning_rate': '9e-05', 'epoch': '0.1847', 'train/total_time_seconds': '65.12', 'train/time_per_step_avg': '0.2378', 'train/epoch_time_elapsed': '78.35', 'train/estimated_remaining_minutes': '0.301'}
25
  {'loss': '8.077', 'grad_norm': '2.317e+08', 'learning_rate': '9e-05', 'epoch': '0.186', 'train/total_time_seconds': '70.12', 'train/time_per_step_avg': '0.2378', 'train/epoch_time_elapsed': '84.44', 'train/estimated_remaining_minutes': '0.3133', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '186.1', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0008926', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.2', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '-3.07e-06', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3262', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.3262', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.2949', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6211', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '1897', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.03638', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '0.6523', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '4.438', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-4.438', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '3.609', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '8.047', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '1877', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.01044', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '0.6484', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '3.797', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-3.797', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '3.422', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '7.219', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '2102', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '0.6133', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '0.8242', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '4.688', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '-0.8867', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '4.688', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '5.574', 'train/tensor_act_model_layers_0_mlp/norm': '1878', 'train/tensor_act_model_layers_0_mlp/mean': '0.5703', 'train/tensor_act_model_layers_0_mlp/std': '0.7188', 'train/tensor_act_model_layers_0_mlp/max_abs': '3.297', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '-0.8711', 'train/tensor_act_model_layers_0_mlp/max': '3.297', 'train/tensor_act_model_layers_0_mlp/range': '4.168', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '1897', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '0.5703', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.7305', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '3.422', 'train/tensor_
26
  {'loss': '8.085', 'grad_norm': '1.646e+08', 'learning_rate': '9e-05', 'epoch': '0.1874', 'train/total_time_seconds': '74.72', 'train/time_per_step_avg': '0.2379', 'train/epoch_time_elapsed': '90.08', 'train/estimated_remaining_minutes': '0.3225'}
27
+ {'loss': '8.09', 'grad_norm': '1.091e+08', 'learning_rate': '9e-05', 'epoch': '0.1887', 'train/total_time_seconds': '79.33', 'train/time_per_step_avg': '0.2381', 'train/epoch_time_elapsed': '95.69', 'train/estimated_remaining_minutes': '0.3305'}
28
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
29
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
30
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
31
+ Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:00<00:00, 15.56it/s]
32
+ 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 2900/3500 [02:05<02:48, 3.56it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From πŸ‘‰v4.50πŸ‘ˆ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
33
+ {'loss': '8.097', 'grad_norm': '7.13e+07', 'learning_rate': '9e-05', 'epoch': '0.1901', 'train/total_time_seconds': '84.42', 'train/time_per_step_avg': '0.2386', 'train/epoch_time_elapsed': '102.1', 'train/estimated_remaining_minutes': '0.3393', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '185.9', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0009193', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.2', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '-6.437e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3555', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.3555', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3242', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6797', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '2680', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.06152', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '0.9219', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '5.625', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-5.625', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '4.625', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '10.25', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '2686', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.01282', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '0.9258', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '5.281', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-5.281', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '4.781', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '10.06', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '8155', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '2.547', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '3.062', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '14.38', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.03122', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '-0.3613', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '14.38', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '14.74', 'train/tensor_act_model_layers_0_mlp/norm': '4345', 'train/tensor_act_model_layers_0_mlp/mean': '1.586', 'train/tensor_act_model_layers_0_mlp/std': '1.414', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '-0.3594', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '4.359', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '4353', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '1.586', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '1.414', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.219', 'train/tensor_act_mod
34
+ {'loss': '8.106', 'grad_norm': '3.185e+07', 'learning_rate': '9e-05', 'epoch': '0.1914', 'train/total_time_seconds': '89.24', 'train/time_per_step_avg': '0.2412', 'train/epoch_time_elapsed': '107.9', 'train/estimated_remaining_minutes': '0.3456'}
35
+ {'loss': '8.104', 'grad_norm': '1.743e+07', 'learning_rate': '9e-05', 'epoch': '0.1928', 'train/total_time_seconds': '94.27', 'train/time_per_step_avg': '0.2415', 'train/epoch_time_elapsed': '114.1', 'train/estimated_remaining_minutes': '0.3516', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '186.2', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.001076', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.5', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '0.0002203', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.0918', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3281', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.3281', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3125', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6406', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '3676', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.06934', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.266', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '6.031', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.031', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '5.938', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '11.97', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '3721', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.01917', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.281', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '6.25', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-6.25', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '6.219', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '12.47', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '2.066e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '6.844', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '7.406', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '32.25', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.2761', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '-0.4883', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '32.25', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '32.74', 'train/tensor_act_model_layers_0_mlp/norm': '5854', 'train/tensor_act_model_layers_0_mlp/mean': '2.469', 'train/tensor_act_model_layers_0_mlp/std': '1.438', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '-0.4863', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '4.486', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '5859', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '2.469', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '1.445', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.281', 'train/tensor_act_model
36
+ {'loss': '8.107', 'grad_norm': '1.226e+07', 'learning_rate': '9e-05', 'epoch': '0.1941', 'train/total_time_seconds': '98.84', 'train/time_per_step_avg': '0.2412', 'train/epoch_time_elapsed': '119.7', 'train/estimated_remaining_minutes': '0.3546'}
37
+ {'loss': '8.105', 'grad_norm': '1.022e+07', 'learning_rate': '9e-05', 'epoch': '0.1955', 'train/total_time_seconds': '103.5', 'train/time_per_step_avg': '0.2413', 'train/epoch_time_elapsed': '125.3', 'train/estimated_remaining_minutes': '0.3568'}
38
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
39
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
40
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
41
+ Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:00<00:00, 17.45it/s]
42
+ 86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 3000/3500 [02:34<02:20, 3.56it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From πŸ‘‰v4.50πŸ‘ˆ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
43
+ {'loss': '8.104', 'grad_norm': '1.016e+07', 'learning_rate': '9e-05', 'epoch': '0.1968', 'train/total_time_seconds': '108.6', 'train/time_per_step_avg': '0.2415', 'train/epoch_time_elapsed': '131.7', 'train/estimated_remaining_minutes': '0.3594', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '185.9', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0009613', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.1', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '7.629e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.332', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.332', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3105', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6426', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '4636', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.08057', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.602', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '7.219', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.562', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '7.219', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '13.78', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '4710', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.02734', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.625', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '7.562', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-7.562', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '7.375', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '14.94', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '4.009e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '14.06', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '13.56', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '60', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.4477', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '-0.08105', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '60', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '60.08', 'train/tensor_act_model_layers_0_mlp/norm': '6960', 'train/tensor_act_model_layers_0_mlp/mean': '3.219', 'train/tensor_act_model_layers_0_mlp/std': '1.078', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '-0.08105', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '4.081', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '6963', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '3.219', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '1.078', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.281', 'train/tensor_act_model_l
44
+ {'loss': '8.105', 'grad_norm': '1.979e+07', 'learning_rate': '9e-05', 'epoch': '0.1982', 'train/total_time_seconds': '113.2', 'train/time_per_step_avg': '0.2395', 'train/epoch_time_elapsed': '137.4', 'train/estimated_remaining_minutes': '0.3593'}
45
+ {'loss': '8.1', 'grad_norm': '6.488e+06', 'learning_rate': '9e-05', 'epoch': '0.1995', 'train/total_time_seconds': '118.3', 'train/time_per_step_avg': '0.2399', 'train/epoch_time_elapsed': '143.6', 'train/estimated_remaining_minutes': '0.3596', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '186', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0009537', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.2', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '2.337e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3242', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.3242', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3125', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6367', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '4988', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.09424', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.719', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '7.188', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.5', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '7.188', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '13.69', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '5069', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.03174', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.75', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '7.75', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-7.5', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '7.75', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '15.25', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '4.896e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '18.12', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '15.62', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '68.5', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.5547', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '0.1484', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '68.5', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '68.35', 'train/tensor_act_model_layers_0_mlp/norm': '7503', 'train/tensor_act_model_layers_0_mlp/mean': '3.594', 'train/tensor_act_model_layers_0_mlp/std': '0.7109', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '0.1484', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '3.852', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '7505', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '3.594', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.7188', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.281', 'train/tensor_act_model_layers_0
46
+ {'loss': '8.102', 'grad_norm': '4.719e+06', 'learning_rate': '9e-05', 'epoch': '0.2009', 'train/total_time_seconds': '122.9', 'train/time_per_step_avg': '0.2403', 'train/epoch_time_elapsed': '149.2', 'train/estimated_remaining_minutes': '0.3573'}
47
+ {'loss': '8.102', 'grad_norm': '3.654e+06', 'learning_rate': '9e-05', 'epoch': '0.2022', 'train/total_time_seconds': '127.5', 'train/time_per_step_avg': '0.2402', 'train/epoch_time_elapsed': '154.8', 'train/estimated_remaining_minutes': '0.3541'}
48
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
49
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
50
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
51
+ Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:00<00:00, 15.23it/s]
52
+ 89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 3100/3500 [03:04<01:51, 3.60it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From πŸ‘‰v4.50οΏ½οΏ½οΏ½ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
53
+ {'loss': '8.106', 'grad_norm': '2.703e+06', 'learning_rate': '9e-05', 'epoch': '0.2036', 'train/total_time_seconds': '132.5', 'train/time_per_step_avg': '0.2397', 'train/epoch_time_elapsed': '161.2', 'train/estimated_remaining_minutes': '0.3511', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '186.1', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.000946', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.3', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '6.58e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3418', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.332', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3418', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6738', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '5243', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.1055', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.805', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '7.375', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.656', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '7.375', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '14.03', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '5329', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.0332', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.836', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '7.719', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-7.5', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '7.719', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '15.22', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '5.638e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '21.88', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '16.75', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '83.5', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.6718', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '0.8672', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '83.5', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '82.63', 'train/tensor_act_model_layers_0_mlp/norm': '7914', 'train/tensor_act_model_layers_0_mlp/mean': '3.844', 'train/tensor_act_model_layers_0_mlp/std': '0.3535', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '0.8555', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '3.145', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '7916', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '3.844', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.3672', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.281', 'train/tensor_act_model_laye
54
+ {'loss': '8.108', 'grad_norm': '1.999e+06', 'learning_rate': '9e-05', 'epoch': '0.2049', 'train/total_time_seconds': '137.1', 'train/time_per_step_avg': '0.2394', 'train/epoch_time_elapsed': '166.8', 'train/estimated_remaining_minutes': '0.3458'}
55
+ {'loss': '8.102', 'grad_norm': '1.516e+06', 'learning_rate': '9e-05', 'epoch': '0.2063', 'train/total_time_seconds': '142.2', 'train/time_per_step_avg': '0.239', 'train/epoch_time_elapsed': '172.9', 'train/estimated_remaining_minutes': '0.3407', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '185.8', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0009766', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.1', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '7.629e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3184', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.3184', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3125', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6309', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '5401', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.1074', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.859', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '7.875', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.906', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '7.875', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '14.78', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '5496', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.03369', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.898', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '8.062', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-7.562', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '8.062', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '15.62', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '6.123e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '24.62', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '16.88', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '94', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.8062', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '1.547', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '94', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '92.45', 'train/tensor_act_model_layers_0_mlp/norm': '8089', 'train/tensor_act_model_layers_0_mlp/mean': '3.953', 'train/tensor_act_model_layers_0_mlp/std': '0.1523', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '1.477', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '2.523', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '8092', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '3.953', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.1787', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.312', 'train/tensor_act_model_layer
56
+ {'loss': '8.101', 'grad_norm': '1.155e+06', 'learning_rate': '9e-05', 'epoch': '0.2076', 'train/total_time_seconds': '146.8', 'train/time_per_step_avg': '0.239', 'train/epoch_time_elapsed': '178.6', 'train/estimated_remaining_minutes': '0.3336'}
57
+ {'loss': '8.105', 'grad_norm': '8.602e+05', 'learning_rate': '9e-05', 'epoch': '0.209', 'train/total_time_seconds': '151.3', 'train/time_per_step_avg': '0.2385', 'train/epoch_time_elapsed': '184.1', 'train/estimated_remaining_minutes': '0.3254'}
58
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
59
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
60
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
61
+ Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:00<00:00, 15.19it/s]
62
+ 91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 3200/3500 [03:40<01:51, 2.68it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From πŸ‘‰v4.50πŸ‘ˆ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
63
+ {'loss': '8.107', 'grad_norm': '6.513e+05', 'learning_rate': '9e-05', 'epoch': '0.2103', 'train/total_time_seconds': '156.4', 'train/time_per_step_avg': '0.2383', 'train/epoch_time_elapsed': '190.5', 'train/estimated_remaining_minutes': '0.3174', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '186', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0009766', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.2', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '-3.91e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3496', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.332', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3496', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6816', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '5413', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.1016', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.867', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '7.594', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.719', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '7.594', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '14.31', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '5512', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.03271', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.906', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '7.938', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-7.531', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '7.938', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '15.47', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '6.279e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '25.75', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '16.62', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '95.5', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.8794', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '2.266', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '95.5', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '93.23', 'train/tensor_act_model_layers_0_mlp/norm': '8143', 'train/tensor_act_model_layers_0_mlp/mean': '3.969', 'train/tensor_act_model_layers_0_mlp/std': '0.0752', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '2.047', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '1.953', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '8145', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '3.969', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.1187', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.344', 'train/tensor_act_model_lay
64
+ {'loss': '8.105', 'grad_norm': '5.018e+05', 'learning_rate': '9e-05', 'epoch': '0.2117', 'train/total_time_seconds': '161', 'train/time_per_step_avg': '0.2391', 'train/epoch_time_elapsed': '196.3', 'train/estimated_remaining_minutes': '0.3077'}
65
+ {'loss': '8.103', 'grad_norm': '4.321e+05', 'learning_rate': '9e-05', 'epoch': '0.213', 'train/total_time_seconds': '167.5', 'train/time_per_step_avg': '0.2533', 'train/epoch_time_elapsed': '204.2', 'train/estimated_remaining_minutes': '0.3003', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '185.9', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.001007', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.1', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '3.886e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3496', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.332', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3496', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6816', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '5445', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.0957', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.875', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '7.594', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.781', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '7.594', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '14.38', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '5545', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.03149', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.914', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '7.969', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-7.5', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '7.969', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '15.47', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '6.421e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '26.88', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '16.25', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '99', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.9393', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '2.984', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '99', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '96.02', 'train/tensor_act_model_layers_0_mlp/norm': '8169', 'train/tensor_act_model_layers_0_mlp/mean': '3.984', 'train/tensor_act_model_layers_0_mlp/std': '0.03857', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '2.531', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '1.469', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '8171', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '3.984', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.09961', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.344', 'train/tensor_act_model_layers_
66
+ {'loss': '8.102', 'grad_norm': '3.359e+05', 'learning_rate': '9e-05', 'epoch': '0.2144', 'train/total_time_seconds': '174.6', 'train/time_per_step_avg': '0.2783', 'train/epoch_time_elapsed': '212.4', 'train/estimated_remaining_minutes': '0.2928'}
67
+ {'loss': '8.105', 'grad_norm': '3.011e+05', 'learning_rate': '9e-05', 'epoch': '0.2157', 'train/total_time_seconds': '181.4', 'train/time_per_step_avg': '0.3008', 'train/epoch_time_elapsed': '220.2', 'train/estimated_remaining_minutes': '0.2834'}
68
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
69
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
70
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
71
+ Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:00<00:00, 13.47it/s]
72
+ 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 3300/3500 [04:22<01:23, 2.41it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From πŸ‘‰v4.50πŸ‘ˆ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
73
+ {'loss': '8.106', 'grad_norm': '2.345e+05', 'learning_rate': '9e-05', 'epoch': '0.2171', 'train/total_time_seconds': '189.4', 'train/time_per_step_avg': '0.3297', 'train/epoch_time_elapsed': '229.5', 'train/estimated_remaining_minutes': '0.2744', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '186.1', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0009766', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.4', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '4.673e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.332', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.332', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.293', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.625', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '5441', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.09473', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.875', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '7.719', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.688', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '7.719', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '14.41', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '5544', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.03174', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.914', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '7.875', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-7.781', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '7.875', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '15.66', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '6.471e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '27.38', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '15.81', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '100.5', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.9718', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '3.797', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '100.5', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '96.7', 'train/tensor_act_model_layers_0_mlp/norm': '8179', 'train/tensor_act_model_layers_0_mlp/mean': '4', 'train/tensor_act_model_layers_0_mlp/std': '0.02234', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '2.953', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '1.047', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '8181', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '4', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.09424', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.281', 'train/tensor_act_model_layers_0
74
+ {'loss': '8.104', 'grad_norm': '2.202e+05', 'learning_rate': '9e-05', 'epoch': '0.2184', 'train/total_time_seconds': '196.2', 'train/time_per_step_avg': '0.3518', 'train/epoch_time_elapsed': '237.4', 'train/estimated_remaining_minutes': '0.2624'}
75
+ {'loss': '8.105', 'grad_norm': '1.956e+05', 'learning_rate': '9e-05', 'epoch': '0.2198', 'train/total_time_seconds': '204.4', 'train/time_per_step_avg': '0.369', 'train/epoch_time_elapsed': '246.7', 'train/estimated_remaining_minutes': '0.2508', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '186.1', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0009232', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.4', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '4.798e-06', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3457', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.3418', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3457', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6875', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '5425', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.09424', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.867', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '7.781', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.75', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '7.781', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '14.53', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '5528', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.03052', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.906', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '8.188', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-7.562', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '8.188', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '15.75', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '6.477e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '27.5', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '15.56', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '106', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.9863', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '4.562', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '106', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '101.4', 'train/tensor_act_model_layers_0_mlp/norm': '8183', 'train/tensor_act_model_layers_0_mlp/mean': '4', 'train/tensor_act_model_layers_0_mlp/std': '0.01434', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '3.266', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '0.7344', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '8185', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '4', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.09277', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.312', 'train/tensor_act_model_layers_0_
76
+ {'loss': '8.105', 'grad_norm': '1.761e+05', 'learning_rate': '9e-05', 'epoch': '0.2211', 'train/total_time_seconds': '211.3', 'train/time_per_step_avg': '0.3675', 'train/epoch_time_elapsed': '254.8', 'train/estimated_remaining_minutes': '0.2363'}
77
+ {'loss': '8.105', 'grad_norm': '1.618e+05', 'learning_rate': '9e-05', 'epoch': '0.2224', 'train/total_time_seconds': '218.1', 'train/time_per_step_avg': '0.3667', 'train/epoch_time_elapsed': '262.5', 'train/estimated_remaining_minutes': '0.2203'}
78
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
79
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
80
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
81
+ Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:00<00:00, 17.62it/s]
82
+ 97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 3400/3500 [05:04<00:41, 2.43it/s][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From πŸ‘‰v4.50πŸ‘ˆ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
83
+ {'loss': '8.104', 'grad_norm': '1.413e+05', 'learning_rate': '9e-05', 'epoch': '0.2238', 'train/total_time_seconds': '226.2', 'train/time_per_step_avg': '0.3683', 'train/epoch_time_elapsed': '271.9', 'train/estimated_remaining_minutes': '0.2044', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '185.9', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0009918', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.1', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '6.39e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3496', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.332', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3496', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6816', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '5434', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.08545', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.875', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '7.438', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.719', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '7.438', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '14.16', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '5538', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.02991', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.914', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '8.25', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-7.25', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '8.25', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '15.5', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '6.497e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '27.75', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '15.31', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '94', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.9914', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '5.25', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '94', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '88.75', 'train/tensor_act_model_layers_0_mlp/norm': '8185', 'train/tensor_act_model_layers_0_mlp/mean': '4', 'train/tensor_act_model_layers_0_mlp/std': '0.01178', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '3.453', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '0.5469', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '8187', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '4', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.09229', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.344', 'train/tensor_act_model_layers_0_residu
84
+ {'loss': '8.104', 'grad_norm': '1.341e+05', 'learning_rate': '9e-05', 'epoch': '0.2251', 'train/total_time_seconds': '232.9', 'train/time_per_step_avg': '0.3666', 'train/epoch_time_elapsed': '279.7', 'train/estimated_remaining_minutes': '0.1859'}
85
+ {'loss': '8.104', 'grad_norm': '1.224e+05', 'learning_rate': '9e-05', 'epoch': '0.2265', 'train/total_time_seconds': '240.5', 'train/time_per_step_avg': '0.3606', 'train/epoch_time_elapsed': '288.8', 'train/estimated_remaining_minutes': '0.167', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '185.7', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0009232', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '-1.317e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3203', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.3203', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.2988', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6191', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '5433', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.08789', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.875', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '7.5', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.688', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '7.5', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '14.19', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '5535', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.02966', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.914', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '7.906', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-7.594', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '7.906', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '15.5', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '6.506e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '28', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '15.06', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '95.5', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.9951', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '5.781', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '95.5', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '89.72', 'train/tensor_act_model_layers_0_mlp/norm': '8187', 'train/tensor_act_model_layers_0_mlp/mean': '4', 'train/tensor_act_model_layers_0_mlp/std': '0.00946', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '3.578', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '0.4219', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '8189', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '4', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.0918', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.312', 'train/tensor_act_model_layers_0_residu
86
+ {'loss': '8.106', 'grad_norm': '1.198e+05', 'learning_rate': '9e-05', 'epoch': '0.2278', 'train/total_time_seconds': '247.3', 'train/time_per_step_avg': '0.3595', 'train/epoch_time_elapsed': '296.7', 'train/estimated_remaining_minutes': '0.1463'}
87
+ {'loss': '8.106', 'grad_norm': '1.157e+05', 'learning_rate': '9e-05', 'epoch': '0.2292', 'train/total_time_seconds': '254.2', 'train/time_per_step_avg': '0.3616', 'train/epoch_time_elapsed': '304.7', 'train/estimated_remaining_minutes': '0.1246'}
88
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
89
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
90
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
91
+ Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:00<00:00, 15.19it/s]
92
+ 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 3500/3500 [06:07<00:00, 2.41i[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From πŸ‘‰v4.50πŸ‘ˆ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
93
+ {'loss': '8.106', 'grad_norm': '1.034e+05', 'learning_rate': '9e-05', 'epoch': '0.2305', 'train/total_time_seconds': '261.8', 'train/time_per_step_avg': '0.3564', 'train/epoch_time_elapsed': '313.6', 'train/estimated_remaining_minutes': '0.1021', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '185.9', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.0009308', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.1', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '-1.061e-05', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.3496', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.3203', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3496', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6699', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '5415', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.08252', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.867', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '7.625', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.688', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '7.625', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '14.31', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '5517', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.02844', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.906', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '7.781', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-7.312', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '7.781', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '15.09', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '6.468e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '27.88', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '14.88', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '100', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.9964', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '6.188', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '100', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '93.81', 'train/tensor_act_model_layers_0_mlp/norm': '8187', 'train/tensor_act_model_layers_0_mlp/mean': '4', 'train/tensor_act_model_layers_0_mlp/std': '0.008362', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '3.656', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '0.3438', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '8189', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '4', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.0918', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.312', 'train/tensor_act_model_layer
94
+ {'loss': '8.105', 'grad_norm': '1.014e+05', 'learning_rate': '9e-05', 'epoch': '0.2319', 'train/total_time_seconds': '268.7', 'train/time_per_step_avg': '0.3582', 'train/epoch_time_elapsed': '321.5', 'train/estimated_remaining_minutes': '0.07811'}
95
+ {'loss': '8.108', 'grad_norm': '9.677e+04', 'learning_rate': '9e-05', 'epoch': '0.2332', 'train/total_time_seconds': '276.8', 'train/time_per_step_avg': '0.3638', 'train/epoch_time_elapsed': '330.8', 'train/estimated_remaining_minutes': '0.05334', 'train/tensor_act_model_layers_0_residual_pre_attn/norm': '186', 'train/tensor_act_model_layers_0_residual_pre_attn/mean': '0.00106', 'train/tensor_act_model_layers_0_residual_pre_attn/std': '0.09082', 'train/tensor_act_model_layers_0_residual_pre_attn/max_abs': '0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_pre_attn/min': '-0.3008', 'train/tensor_act_model_layers_0_residual_pre_attn/max': '0.2852', 'train/tensor_act_model_layers_0_residual_pre_attn/range': '0.5859', 'train/tensor_act_model_layers_0_residual_post_attn/norm': '187.3', 'train/tensor_act_model_layers_0_residual_post_attn/mean': '0.0001574', 'train/tensor_act_model_layers_0_residual_post_attn/std': '0.09131', 'train/tensor_act_model_layers_0_residual_post_attn/max_abs': '0.332', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_residual_post_attn/min': '-0.332', 'train/tensor_act_model_layers_0_residual_post_attn/max': '0.3047', 'train/tensor_act_model_layers_0_residual_post_attn/range': '0.6367', 'train/tensor_act_model_layers_0_mlp_gate_proj/norm': '5389', 'train/tensor_act_model_layers_0_mlp_gate_proj/mean': '0.08203', 'train/tensor_act_model_layers_0_mlp_gate_proj/std': '1.859', 'train/tensor_act_model_layers_0_mlp_gate_proj/max_abs': '7.469', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_gate_proj/min': '-6.688', 'train/tensor_act_model_layers_0_mlp_gate_proj/max': '7.469', 'train/tensor_act_model_layers_0_mlp_gate_proj/range': '14.16', 'train/tensor_act_model_layers_0_mlp_up_proj/norm': '5488', 'train/tensor_act_model_layers_0_mlp_up_proj/mean': '0.02856', 'train/tensor_act_model_layers_0_mlp_up_proj/std': '1.898', 'train/tensor_act_model_layers_0_mlp_up_proj/max_abs': '7.906', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp_up_proj/min': '-7.531', 'train/tensor_act_model_layers_0_mlp_up_proj/max': '7.906', 'train/tensor_act_model_layers_0_mlp_up_proj/range': '15.44', 'train/tensor_act_model_layers_0_mlp_down_proj/norm': '6.424e+04', 'train/tensor_act_model_layers_0_mlp_down_proj/mean': '27.75', 'train/tensor_act_model_layers_0_mlp_down_proj/std': '14.69', 'train/tensor_act_model_layers_0_mlp_down_proj/max_abs': '99.5', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit': '0.9977', 'train/tensor_act_model_layers_0_mlp_down_proj/min': '6.562', 'train/tensor_act_model_layers_0_mlp_down_proj/max': '99.5', 'train/tensor_act_model_layers_0_mlp_down_proj/range': '92.94', 'train/tensor_act_model_layers_0_mlp/norm': '8188', 'train/tensor_act_model_layers_0_mlp/mean': '4', 'train/tensor_act_model_layers_0_mlp/std': '0.007172', 'train/tensor_act_model_layers_0_mlp/max_abs': '4', 'train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit': '0', 'train/tensor_act_model_layers_0_mlp/frac_near_user_limit': '0', 'train/tensor_act_model_layers_0_mlp/min': '3.703', 'train/tensor_act_model_layers_0_mlp/max': '4', 'train/tensor_act_model_layers_0_mlp/range': '0.2969', 'train/tensor_act_model_layers_0_residual_post_mlp/norm': '8190', 'train/tensor_act_model_layers_0_residual_post_mlp/mean': '4', 'train/tensor_act_model_layers_0_residual_post_mlp/std': '0.0918', 'train/tensor_act_model_layers_0_residual_post_mlp/max_abs': '4.312', 'train/tensor_act_model_layers_0_
96
+ {'loss': '8.108', 'grad_norm': '9.779e+04', 'learning_rate': '9e-05', 'epoch': '0.2346', 'train/total_time_seconds': '283.8', 'train/time_per_step_avg': '0.365', 'train/epoch_time_elapsed': '338.8', 'train/estimated_remaining_minutes': '0.02718'}
97
+ {'loss': '8.109', 'grad_norm': '8.653e+04', 'learning_rate': '9e-05', 'epoch': '0.2359', 'train/total_time_seconds': '290.7', 'train/time_per_step_avg': '0.3643', 'train/epoch_time_elapsed': '346.6', 'train/estimated_remaining_minutes': '0'}
98
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
99
+ {'eval_loss': '8.104', 'eval_runtime': '20.59', 'eval_samples_per_second': '462.8', 'eval_steps_per_second': '0.486', 'epoch': '0.2359', 'train/total_time_seconds': '290.7', 'train/time_per_step_avg': '0.3643', 'train/epoch_time_elapsed': '367.2', 'train/estimated_remaining_minutes': '0'}
100
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
101
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
102
+ Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:00<00:00, 2.60it/s]
103
+ 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 3500/3500 [06:07<00:00, 9.52it/s]0:00, 2.60it/s]
104
+ {'train_runtime': '368.5', 'train_samples_per_second': '607.8', 'train_steps_per_second': '9.497', 'train_loss': '2.204', 'epoch': '0.2359', 'train/total_time_seconds': '290.7', 'train/time_per_step_avg': '0.3643', 'train/epoch_time_elapsed': '367.7', 'train/estimated_remaining_minutes': '0'}
105
+ 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 10/10 [00:18<00:00, 1.84s/it]
106
+ [transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From πŸ‘‰v4.50πŸ‘ˆ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
107
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
108
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
109
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
110
+ Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:00<00:00, 15.55it/s]
111
+ >>> FINISHED glu-silu-waleed10-100L successfully