backup: training logs before server expiry
Browse files- .gitattributes +7 -0
- logs_backup/diag_mem_bs4.log +0 -0
- logs_backup/star_atc_4gpu_fixed.log +0 -0
- logs_backup/star_atc_8gpu.log +3 -0
- logs_backup/star_atc_adaptive.log +0 -0
- logs_backup/star_atc_augmented.log +0 -0
- logs_backup/star_atc_best.log +0 -0
- logs_backup/star_atc_best_augmented.log +0 -0
- logs_backup/star_atc_k1L100.log +0 -0
- logs_backup/star_atc_k1L100_adaptive.log +3 -0
- logs_backup/star_atc_twoptr_repro.log +0 -0
- logs_backup/star_coconut_4gpu.log +3 -0
- logs_backup/star_coconut_k10L10_fixed.log +3 -0
- logs_backup/star_datacurr_4gpu_fixed.log +0 -0
- logs_backup/star_h2h_atc.log +0 -0
- logs_backup/star_h2h_nocot.log +0 -0
- logs_backup/star_k5_adaptive.log +0 -0
- logs_backup/star_k5_augmented.log +0 -0
- logs_backup/star_k5_no_thought.log +0 -0
- logs_backup/star_lr_1e-4.log +0 -0
- logs_backup/star_lr_1e-5.log +0 -0
- logs_backup/star_lr_2e-5.log +0 -0
- logs_backup/star_lr_5e-5.log +0 -0
- logs_backup/star_n1000_adaptive.log +0 -0
- logs_backup/star_n200_adaptive.log +0 -0
- logs_backup/star_n200_augmented.log +0 -0
- logs_backup/star_no_cot_repro.log +0 -0
- logs_backup/star_no_thought.log +0 -0
- logs_backup/star_nocot_4gpu_fixed.log +0 -0
- logs_backup/star_nocot_dc_k1L100.log +3 -0
- logs_backup/star_nocot_dc_lr1e5.log +0 -0
- logs_backup/star_nocot_nocurr_4gpu.log +3 -0
- logs_backup/star_nocot_nocurr_k1L100.log +3 -0
- logs_backup/star_nocot_nocurr_lr1e5.log +0 -0
- logs_backup/star_smoke.log +119 -0
- logs_backup/star_sweep_bs128_lr1e-4.log +296 -0
- logs_backup/star_sweep_bs128_lr1e-5.log +296 -0
- logs_backup/star_sweep_bs256_lr1e-4.log +290 -0
- logs_backup/star_sweep_bs256_lr1e-5.log +290 -0
- logs_backup/star_sweep_bs32_lr1e-4.log +152 -0
- logs_backup/star_sweep_bs32_lr1e-5.log +152 -0
- logs_backup/star_sweep_bs64_lr1e-4.log +285 -0
- logs_backup/star_sweep_bs64_lr1e-5.log +284 -0
- logs_backup/star_thoughtformer.log +0 -0
.gitattributes
CHANGED
|
@@ -902,3 +902,10 @@ checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_n
|
|
| 902 |
checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_no-reset_epoch-334c5641/train_state_30 filter=lfs diff=lfs merge=lfs -text
|
| 903 |
checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_no-reset_epoch-334c5641/checkpoint_31 filter=lfs diff=lfs merge=lfs -text
|
| 904 |
checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_no-reset_epoch-334c5641/train_state_31 filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 902 |
checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_no-reset_epoch-334c5641/train_state_30 filter=lfs diff=lfs merge=lfs -text
|
| 903 |
checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_no-reset_epoch-334c5641/checkpoint_31 filter=lfs diff=lfs merge=lfs -text
|
| 904 |
checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_no-reset_epoch-334c5641/train_state_31 filter=lfs diff=lfs merge=lfs -text
|
| 905 |
+
logs_backup/star_atc_8gpu.log filter=lfs diff=lfs merge=lfs -text
|
| 906 |
+
logs_backup/star_atc_k1L100_adaptive.log filter=lfs diff=lfs merge=lfs -text
|
| 907 |
+
logs_backup/star_coconut_4gpu.log filter=lfs diff=lfs merge=lfs -text
|
| 908 |
+
logs_backup/star_coconut_k10L10_fixed.log filter=lfs diff=lfs merge=lfs -text
|
| 909 |
+
logs_backup/star_nocot_dc_k1L100.log filter=lfs diff=lfs merge=lfs -text
|
| 910 |
+
logs_backup/star_nocot_nocurr_4gpu.log filter=lfs diff=lfs merge=lfs -text
|
| 911 |
+
logs_backup/star_nocot_nocurr_k1L100.log filter=lfs diff=lfs merge=lfs -text
|
logs_backup/diag_mem_bs4.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_atc_4gpu_fixed.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_atc_8gpu.log
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a895d341afb2c0c7e26ff3ff913350a1fb05fca0de77269330548c9de7512272
|
| 3 |
+
size 24023972
|
logs_backup/star_atc_adaptive.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_atc_augmented.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_atc_best.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_atc_best_augmented.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_atc_k1L100.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_atc_k1L100_adaptive.log
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:806b91d45db3b67ae77b3133beada72d506f3cb90816776af0af70b8d12d23a2
|
| 3 |
+
size 14951470
|
logs_backup/star_atc_twoptr_repro.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_coconut_4gpu.log
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5e5e5cc30c7f5a697b9d5d19172900cf4b90c8b0a028a6c90983761dc9f474b4
|
| 3 |
+
size 17755610
|
logs_backup/star_coconut_k10L10_fixed.log
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:08f7989e32e077f0c04fa2fea732fdf98c5b75501d4ea9cc8f3ba2c5614cc178
|
| 3 |
+
size 17668432
|
logs_backup/star_datacurr_4gpu_fixed.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_h2h_atc.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_h2h_nocot.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_k5_adaptive.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_k5_augmented.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_k5_no_thought.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_lr_1e-4.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_lr_1e-5.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_lr_2e-5.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_lr_5e-5.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_n1000_adaptive.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_n200_adaptive.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_n200_augmented.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_no_cot_repro.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_no_thought.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_nocot_4gpu_fixed.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_nocot_dc_k1L100.log
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:74e40e7fd0a0e606cc794c400df81a09a622de74aff7cfd02e6a4c9bab495446
|
| 3 |
+
size 13442075
|
logs_backup/star_nocot_dc_lr1e5.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_nocot_nocurr_4gpu.log
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2357a9bccacff1e8e7a7a86e87fa310af6bf47c3369a6ead5a2929c538639eb8
|
| 3 |
+
size 37777886
|
logs_backup/star_nocot_nocurr_k1L100.log
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f30faeed2612df03f87580ae5d59b74c1bf76230bbdfd4c384117107438d2121
|
| 3 |
+
size 24073355
|
logs_backup/star_nocot_nocurr_lr1e5.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
logs_backup/star_smoke.log
ADDED
|
@@ -0,0 +1,119 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Config: {'project': 'thoughtformer', 'name': 'star-smoke', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': False, 'staging': 'two_pointers', 'init_thought_stage': 0, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 0, 'max_latent_stage': 0, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 4, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 2, 'lr': 0.0001, 'weight_decay': 0.01, 'group': 'smoke', 'tags': ['star', 'k10', 'L10', 'no_cot'], 'run_type': 'pilot'}
|
| 2 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 3 |
+
return func(*args, **kwargs)
|
| 4 |
+
[rank0]:[W528 07:02:16.327385348 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 5 |
+
|
| 6 |
+
Running FSDP on rank = 0, world size = 1
|
| 7 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
|
| 8 |
+
_init_core_state(
|
| 9 |
+
FullyShardedDataParallel(
|
| 10 |
+
(_fsdp_wrapped_module): Coconut(
|
| 11 |
+
(base_causallm): Qwen3ForCausalLM(
|
| 12 |
+
(model): Qwen3Model(
|
| 13 |
+
(embed_tokens): Embedding(151672, 1024)
|
| 14 |
+
(layers): ModuleList(
|
| 15 |
+
(0-27): 28 x FullyShardedDataParallel(
|
| 16 |
+
(_fsdp_wrapped_module): Qwen3DecoderLayer(
|
| 17 |
+
(self_attn): Qwen3Attention(
|
| 18 |
+
(q_proj): Linear(in_features=1024, out_features=2048, bias=False)
|
| 19 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 20 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 21 |
+
(o_proj): Linear(in_features=2048, out_features=1024, bias=False)
|
| 22 |
+
(q_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 23 |
+
(k_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 24 |
+
)
|
| 25 |
+
(mlp): Qwen3MLP(
|
| 26 |
+
(gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 27 |
+
(up_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 28 |
+
(down_proj): Linear(in_features=3072, out_features=1024, bias=False)
|
| 29 |
+
(act_fn): SiLUActivation()
|
| 30 |
+
)
|
| 31 |
+
(input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 32 |
+
(post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 33 |
+
)
|
| 34 |
+
)
|
| 35 |
+
)
|
| 36 |
+
(norm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 37 |
+
(rotary_emb): Qwen3RotaryEmbedding()
|
| 38 |
+
)
|
| 39 |
+
(lm_head): Linear(in_features=1024, out_features=151936, bias=False)
|
| 40 |
+
)
|
| 41 |
+
(embedding): Embedding(151672, 1024)
|
| 42 |
+
)
|
| 43 |
+
)
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
|
| 47 |
+
wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 48 |
+
wandb: setting up run kbjpqcuk
|
| 49 |
+
wandb: Tracking run with wandb version 0.25.1
|
| 50 |
+
wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260528_070253-kbjpqcuk
|
| 51 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 52 |
+
wandb: Syncing run star-smoke_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_no-tc_two_pointers-bdad6776
|
| 53 |
+
wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
|
| 54 |
+
wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/kbjpqcuk
|
| 55 |
+
|
| 56 |
+
============================================================
|
| 57 |
+
EPOCH 0/2 (thought_stage=0, data_stage=1)
|
| 58 |
+
============================================================
|
| 59 |
+
thought_stage=0, c_thought=0, max_difficulty=1
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
|
| 65 |
+
LR schedule: cosine, warmup=125 steps, total=5000 steps (max_steps/epoch=2500)
|
| 66 |
+
|
| 67 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/autograd/graph.py:882: UserWarning: The AccumulateGrad node's stream does not match the stream of the node that produced the incoming gradient. This may incur unnecessary synchronization and break CUDA graph capture if the AccumulateGrad node's stream is the default stream. This mismatch is caused by an AccumulateGrad node created prior to the current iteration being kept alive. This can happen if the autograd graph is still being kept alive by tensors such as the loss, or if you are using DDP, which will stash a reference to the node. To resolve the mismatch, delete all references to the autograd graph or ensure that DDP initialization is performed under the same stream as subsequent forwards. If the mismatch is intentional, you can use torch.autograd.graph.set_warn_on_accumulate_grad_stream_mismatch(False) to suppress this warning. (Triggered internally at /pytorch/torch/csrc/autograd/input_buffer.cpp:240.)
|
| 68 |
+
return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
|
| 69 |
+
rank 0 | gpu 0 | alloc=6.35 GB | reserved=27.88 GB | max=21.84 GB
|
| 70 |
+
|
| 71 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 72 |
+
return func(*args, **kwargs)
|
| 73 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:821: FutureWarning: FSDP.state_dict_type() and FSDP.set_state_dict_type() are being deprecated. Please use APIs, get_state_dict() and set_state_dict(), which can support different parallelisms, FSDP1, FSDP2, DDP. API doc: https://pytorch.org/docs/stable/distributed.checkpoint.html#torch.distributed.checkpoint.state_dict.get_state_dict .Tutorial: https://pytorch.org/tutorials/recipes/distributed_checkpoint_recipe.html .
|
| 74 |
+
prev_state_dict_settings = FullyShardedDataParallel.set_state_dict_type(
|
| 75 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/utils/_contextlib.py:124: UserWarning: When using ``NO_SHARD`` for ``ShardingStrategy``, full_state_dict will be returned.
|
| 76 |
+
return func(*args, **kwargs)
|
| 77 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/_optim_utils.py:1172: UserWarning: `_get_pg_default_device` will be deprecated, it only stays for backward-compatibility reason. If you need to find a device for object collectives, please use `_get_object_coll_device`. If you need to query the device types supported by group, please use `_device_capability(group)`.
|
| 78 |
+
device = _get_pg_default_device(group)
|
| 79 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:828: FutureWarning: FSDP.state_dict_type() and FSDP.set_state_dict_type() are being deprecated. Please use APIs, get_state_dict() and set_state_dict(), which can support different parallelisms, FSDP1, FSDP2, DDP. API doc: https://pytorch.org/docs/stable/distributed.checkpoint.html#torch.distributed.checkpoint.state_dict.get_state_dict .Tutorial: https://pytorch.org/tutorials/recipes/distributed_checkpoint_recipe.html .
|
| 80 |
+
FullyShardedDataParallel.set_state_dict_type(
|
| 81 |
+
saving model + train state.
|
| 82 |
+
Warning: failed to upload checkpoint_1 to HF: (Request ID: Root=1-6a17e919-1dce5dca1d2ddb050a1a0a4f;c5ffd1eb-6364-4939-8651-3bc2bb1d3fe2)
|
| 83 |
+
|
| 84 |
+
403 Forbidden: You need to setup automatic credit recharge in order to upload more data. You can do so at /settings/billing..
|
| 85 |
+
Cannot access content at: https://huggingface.co/seyedparsa/thoughtformer-checkpoints.git/info/lfs/objects/batch.
|
| 86 |
+
Make sure your token has the correct permissions.
|
| 87 |
+
Warning: failed to upload train_state_1 to HF: (Request ID: Root=1-6a17e91d-3685080228a0e1cb2da99c67;496f22d9-c5af-4e21-bb1b-4dc6beb52df0)
|
| 88 |
+
|
| 89 |
+
403 Forbidden: You need to setup automatic credit recharge in order to upload more data. You can do so at /settings/billing..
|
| 90 |
+
Cannot access content at: https://huggingface.co/seyedparsa/thoughtformer-checkpoints.git/info/lfs/objects/batch.
|
| 91 |
+
Make sure your token has the correct permissions.
|
| 92 |
+
Keeping all local checkpoints because HF upload failed; disk fallback until HF recovers.
|
| 93 |
+
eval loss 0.896042138338089
|
| 94 |
+
|
| 95 |
+
[WRONG] Example 0 (diff=3)
|
| 96 |
+
Expected : 'Patrick'
|
| 97 |
+
Extracted: 'Nancy'
|
| 98 |
+
Generated: '### Nancy'
|
| 99 |
+
|
| 100 |
+
[CORRECT] Example 1 (diff=2)
|
| 101 |
+
Expected : 'Roman'
|
| 102 |
+
Extracted: 'Roman'
|
| 103 |
+
Generated: '### Roman'
|
| 104 |
+
|
| 105 |
+
[WRONG] Example 2 (diff=2)
|
| 106 |
+
Expected : 'Harold'
|
| 107 |
+
Extracted: 'Stella'
|
| 108 |
+
Generated: '### Stella'
|
| 109 |
+
|
| 110 |
+
[WRONG] Example 3 (diff=2)
|
| 111 |
+
Expected : 'Larry'
|
| 112 |
+
Extracted: 'Liam'
|
| 113 |
+
Generated: '### Liam'
|
| 114 |
+
|
| 115 |
+
[CORRECT] Example 4 (diff=2)
|
| 116 |
+
Expected : 'Israel'
|
| 117 |
+
Extracted: 'Israel'
|
| 118 |
+
Generated: '### Israel'
|
| 119 |
+
|
logs_backup/star_sweep_bs128_lr1e-4.log
ADDED
|
@@ -0,0 +1,296 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs128-lr1e-4', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 128, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 0.0001, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
|
| 2 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 3 |
+
return func(*args, **kwargs)
|
| 4 |
+
[rank0]:[W530 03:31:55.296639842 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 5 |
+
|
| 6 |
+
Running FSDP on rank = 0, world size = 1
|
| 7 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
|
| 8 |
+
_init_core_state(
|
| 9 |
+
FullyShardedDataParallel(
|
| 10 |
+
(_fsdp_wrapped_module): Coconut(
|
| 11 |
+
(base_causallm): Qwen3ForCausalLM(
|
| 12 |
+
(model): Qwen3Model(
|
| 13 |
+
(embed_tokens): Embedding(151672, 1024)
|
| 14 |
+
(layers): ModuleList(
|
| 15 |
+
(0-27): 28 x FullyShardedDataParallel(
|
| 16 |
+
(_fsdp_wrapped_module): Qwen3DecoderLayer(
|
| 17 |
+
(self_attn): Qwen3Attention(
|
| 18 |
+
(q_proj): Linear(in_features=1024, out_features=2048, bias=False)
|
| 19 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 20 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 21 |
+
(o_proj): Linear(in_features=2048, out_features=1024, bias=False)
|
| 22 |
+
(q_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 23 |
+
(k_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 24 |
+
)
|
| 25 |
+
(mlp): Qwen3MLP(
|
| 26 |
+
(gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 27 |
+
(up_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 28 |
+
(down_proj): Linear(in_features=3072, out_features=1024, bias=False)
|
| 29 |
+
(act_fn): SiLUActivation()
|
| 30 |
+
)
|
| 31 |
+
(input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 32 |
+
(post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 33 |
+
)
|
| 34 |
+
)
|
| 35 |
+
)
|
| 36 |
+
(norm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 37 |
+
(rotary_emb): Qwen3RotaryEmbedding()
|
| 38 |
+
)
|
| 39 |
+
(lm_head): Linear(in_features=1024, out_features=151936, bias=False)
|
| 40 |
+
)
|
| 41 |
+
(embedding): Embedding(151672, 1024)
|
| 42 |
+
)
|
| 43 |
+
)
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
|
| 47 |
+
wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 48 |
+
wandb: Tracking run with wandb version 0.25.1
|
| 49 |
+
wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033222-2vtfied7
|
| 50 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 51 |
+
wandb: Syncing run star-sweep-bs128-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-6c6fcb83
|
| 52 |
+
wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
|
| 53 |
+
wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/2vtfied7
|
| 54 |
+
|
| 55 |
+
============================================================
|
| 56 |
+
EPOCH 0/30 (thought_stage=1, data_stage=1)
|
| 57 |
+
============================================================
|
| 58 |
+
thought_stage=1, c_thought=1, max_difficulty=1
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
LR schedule: cosine, warmup=3 steps, total=2340 steps (max_steps/epoch=78)
|
| 65 |
+
|
| 66 |
+
wandb: WARNING Serializing object of type str that is 1676601 bytes
|
| 67 |
+
Traceback (most recent call last):
|
| 68 |
+
File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 69 |
+
main()
|
| 70 |
+
File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 71 |
+
outputs = parallel_model(**batch)
|
| 72 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 73 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 74 |
+
return self._call_impl(*args, **kwargs)
|
| 75 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 76 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 77 |
+
return forward_call(*args, **kwargs)
|
| 78 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 79 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 80 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 81 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 82 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 83 |
+
return self._call_impl(*args, **kwargs)
|
| 84 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 85 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 86 |
+
return forward_call(*args, **kwargs)
|
| 87 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 88 |
+
File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
|
| 89 |
+
outputs = self.base_causallm(
|
| 90 |
+
^^^^^^^^^^^^^^^^^^^
|
| 91 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 92 |
+
return self._call_impl(*args, **kwargs)
|
| 93 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 94 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 95 |
+
return forward_call(*args, **kwargs)
|
| 96 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 97 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
|
| 98 |
+
output = func(self, *args, **kwargs)
|
| 99 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 100 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
|
| 101 |
+
outputs: BaseModelOutputWithPast = self.model(
|
| 102 |
+
^^^^^^^^^^^
|
| 103 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 104 |
+
return self._call_impl(*args, **kwargs)
|
| 105 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 106 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 107 |
+
return forward_call(*args, **kwargs)
|
| 108 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 109 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
|
| 110 |
+
output = func(self, *args, **kwargs)
|
| 111 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 112 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
|
| 113 |
+
outputs = func(self, *args, **kwargs)
|
| 114 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 115 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
|
| 116 |
+
hidden_states = decoder_layer(
|
| 117 |
+
^^^^^^^^^^^^^^
|
| 118 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 119 |
+
return self._call_impl(*args, **kwargs)
|
| 120 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 121 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 122 |
+
return forward_call(*args, **kwargs)
|
| 123 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 124 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 125 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 126 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 127 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
|
| 128 |
+
return super().__call__(*args, **kwargs)
|
| 129 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 130 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 131 |
+
return self._call_impl(*args, **kwargs)
|
| 132 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 133 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 134 |
+
return inner()
|
| 135 |
+
^^^^^^^
|
| 136 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 137 |
+
result = forward_call(*args, **kwargs)
|
| 138 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 139 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
|
| 140 |
+
hidden_states = self.mlp(hidden_states)
|
| 141 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 142 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 143 |
+
return self._call_impl(*args, **kwargs)
|
| 144 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 145 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 146 |
+
return forward_call(*args, **kwargs)
|
| 147 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 148 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
|
| 149 |
+
down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 150 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 151 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 152 |
+
return self._call_impl(*args, **kwargs)
|
| 153 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 154 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 155 |
+
return forward_call(*args, **kwargs)
|
| 156 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 157 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/activations.py", line 103, in forward
|
| 158 |
+
return nn.functional.silu(input)
|
| 159 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 160 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/functional.py", line 2397, in silu
|
| 161 |
+
return torch._C._nn.silu(input)
|
| 162 |
+
^^^^^^^^^^^^^^^^^^^^^^^^
|
| 163 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 1.19 GiB. GPU 0 has a total capacity of 139.80 GiB of which 583.06 MiB is free. Including non-PyTorch memory, this process has 139.22 GiB memory in use. Of the allocated memory 136.99 GiB is allocated by PyTorch, and 1.03 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 164 |
+
[rank0]: Traceback (most recent call last):
|
| 165 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 166 |
+
[rank0]: main()
|
| 167 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 168 |
+
[rank0]: outputs = parallel_model(**batch)
|
| 169 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 170 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 171 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 172 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 173 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 174 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 175 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 176 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 177 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 178 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 179 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 180 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 181 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 182 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 183 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 184 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 185 |
+
[rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
|
| 186 |
+
[rank0]: outputs = self.base_causallm(
|
| 187 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^
|
| 188 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 189 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 190 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 191 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 192 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 193 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 194 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
|
| 195 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 196 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 197 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
|
| 198 |
+
[rank0]: outputs: BaseModelOutputWithPast = self.model(
|
| 199 |
+
[rank0]: ^^^^^^^^^^^
|
| 200 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 201 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 202 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 203 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 204 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 205 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 206 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
|
| 207 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 208 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 209 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
|
| 210 |
+
[rank0]: outputs = func(self, *args, **kwargs)
|
| 211 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 212 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
|
| 213 |
+
[rank0]: hidden_states = decoder_layer(
|
| 214 |
+
[rank0]: ^^^^^^^^^^^^^^
|
| 215 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 216 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 217 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 218 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 219 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 220 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 221 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 222 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 223 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 224 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
|
| 225 |
+
[rank0]: return super().__call__(*args, **kwargs)
|
| 226 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 227 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 228 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 229 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 230 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 231 |
+
[rank0]: return inner()
|
| 232 |
+
[rank0]: ^^^^^^^
|
| 233 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 234 |
+
[rank0]: result = forward_call(*args, **kwargs)
|
| 235 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 236 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
|
| 237 |
+
[rank0]: hidden_states = self.mlp(hidden_states)
|
| 238 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 239 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 240 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 241 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 242 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 243 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 244 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 245 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
|
| 246 |
+
[rank0]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 247 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 248 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 249 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 250 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 251 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 252 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 253 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 254 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/activations.py", line 103, in forward
|
| 255 |
+
[rank0]: return nn.functional.silu(input)
|
| 256 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 257 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/functional.py", line 2397, in silu
|
| 258 |
+
[rank0]: return torch._C._nn.silu(input)
|
| 259 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^
|
| 260 |
+
[rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 1.19 GiB. GPU 0 has a total capacity of 139.80 GiB of which 583.06 MiB is free. Including non-PyTorch memory, this process has 139.22 GiB memory in use. Of the allocated memory 136.99 GiB is allocated by PyTorch, and 1.03 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 261 |
+
[1;34mwandb[0m:
|
| 262 |
+
[1;34mwandb[0m: 🚀 View run [33mstar-sweep-bs128-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-6c6fcb83[0m at: [34mhttps://wandb.ai/seyedparsa/thoughtformer/runs/2vtfied7[0m
|
| 263 |
+
[1;34mwandb[0m: Find logs at: [1;35mwandb/run-20260530_033222-2vtfied7/logs[0m
|
| 264 |
+
E0530 03:32:33.929000 465640 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 466218) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
|
| 265 |
+
Traceback (most recent call last):
|
| 266 |
+
File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
|
| 267 |
+
sys.exit(main())
|
| 268 |
+
^^^^^^
|
| 269 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
|
| 270 |
+
return f(*args, **kwargs)
|
| 271 |
+
^^^^^^^^^^^^^^^^^^
|
| 272 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
|
| 273 |
+
run(args)
|
| 274 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
|
| 275 |
+
elastic_launch(
|
| 276 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
|
| 277 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 278 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 279 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
|
| 280 |
+
raise ChildFailedError(
|
| 281 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 282 |
+
============================================================
|
| 283 |
+
run.py FAILED
|
| 284 |
+
------------------------------------------------------------
|
| 285 |
+
Failures:
|
| 286 |
+
<NO_OTHER_FAILURES>
|
| 287 |
+
------------------------------------------------------------
|
| 288 |
+
Root Cause (first observed failure):
|
| 289 |
+
[0]:
|
| 290 |
+
time : 2026-05-30_03:32:33
|
| 291 |
+
host : ip-172-31-10-226.us-east-2.compute.internal
|
| 292 |
+
rank : 0 (local_rank: 0)
|
| 293 |
+
exitcode : 1 (pid: 466218)
|
| 294 |
+
error_file: <N/A>
|
| 295 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 296 |
+
============================================================
|
logs_backup/star_sweep_bs128_lr1e-5.log
ADDED
|
@@ -0,0 +1,296 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs128-lr1e-5', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 128, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 1e-05, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
|
| 2 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 3 |
+
return func(*args, **kwargs)
|
| 4 |
+
[rank0]:[W530 03:31:53.943261833 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 5 |
+
|
| 6 |
+
Running FSDP on rank = 0, world size = 1
|
| 7 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
|
| 8 |
+
_init_core_state(
|
| 9 |
+
FullyShardedDataParallel(
|
| 10 |
+
(_fsdp_wrapped_module): Coconut(
|
| 11 |
+
(base_causallm): Qwen3ForCausalLM(
|
| 12 |
+
(model): Qwen3Model(
|
| 13 |
+
(embed_tokens): Embedding(151672, 1024)
|
| 14 |
+
(layers): ModuleList(
|
| 15 |
+
(0-27): 28 x FullyShardedDataParallel(
|
| 16 |
+
(_fsdp_wrapped_module): Qwen3DecoderLayer(
|
| 17 |
+
(self_attn): Qwen3Attention(
|
| 18 |
+
(q_proj): Linear(in_features=1024, out_features=2048, bias=False)
|
| 19 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 20 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 21 |
+
(o_proj): Linear(in_features=2048, out_features=1024, bias=False)
|
| 22 |
+
(q_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 23 |
+
(k_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 24 |
+
)
|
| 25 |
+
(mlp): Qwen3MLP(
|
| 26 |
+
(gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 27 |
+
(up_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 28 |
+
(down_proj): Linear(in_features=3072, out_features=1024, bias=False)
|
| 29 |
+
(act_fn): SiLUActivation()
|
| 30 |
+
)
|
| 31 |
+
(input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 32 |
+
(post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 33 |
+
)
|
| 34 |
+
)
|
| 35 |
+
)
|
| 36 |
+
(norm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 37 |
+
(rotary_emb): Qwen3RotaryEmbedding()
|
| 38 |
+
)
|
| 39 |
+
(lm_head): Linear(in_features=1024, out_features=151936, bias=False)
|
| 40 |
+
)
|
| 41 |
+
(embedding): Embedding(151672, 1024)
|
| 42 |
+
)
|
| 43 |
+
)
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
|
| 47 |
+
wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 48 |
+
wandb: Tracking run with wandb version 0.25.1
|
| 49 |
+
wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033220-efpd3uen
|
| 50 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 51 |
+
wandb: Syncing run star-sweep-bs128-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-46d2d3b7
|
| 52 |
+
wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
|
| 53 |
+
wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/efpd3uen
|
| 54 |
+
|
| 55 |
+
============================================================
|
| 56 |
+
EPOCH 0/30 (thought_stage=1, data_stage=1)
|
| 57 |
+
============================================================
|
| 58 |
+
thought_stage=1, c_thought=1, max_difficulty=1
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
LR schedule: cosine, warmup=3 steps, total=2340 steps (max_steps/epoch=78)
|
| 65 |
+
|
| 66 |
+
wandb: WARNING Serializing object of type str that is 1676601 bytes
|
| 67 |
+
Traceback (most recent call last):
|
| 68 |
+
File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 69 |
+
main()
|
| 70 |
+
File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 71 |
+
outputs = parallel_model(**batch)
|
| 72 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 73 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 74 |
+
return self._call_impl(*args, **kwargs)
|
| 75 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 76 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 77 |
+
return forward_call(*args, **kwargs)
|
| 78 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 79 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 80 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 81 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 82 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 83 |
+
return self._call_impl(*args, **kwargs)
|
| 84 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 85 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 86 |
+
return forward_call(*args, **kwargs)
|
| 87 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 88 |
+
File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
|
| 89 |
+
outputs = self.base_causallm(
|
| 90 |
+
^^^^^^^^^^^^^^^^^^^
|
| 91 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 92 |
+
return self._call_impl(*args, **kwargs)
|
| 93 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 94 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 95 |
+
return forward_call(*args, **kwargs)
|
| 96 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 97 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
|
| 98 |
+
output = func(self, *args, **kwargs)
|
| 99 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 100 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
|
| 101 |
+
outputs: BaseModelOutputWithPast = self.model(
|
| 102 |
+
^^^^^^^^^^^
|
| 103 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 104 |
+
return self._call_impl(*args, **kwargs)
|
| 105 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 106 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 107 |
+
return forward_call(*args, **kwargs)
|
| 108 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 109 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
|
| 110 |
+
output = func(self, *args, **kwargs)
|
| 111 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 112 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
|
| 113 |
+
outputs = func(self, *args, **kwargs)
|
| 114 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 115 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
|
| 116 |
+
hidden_states = decoder_layer(
|
| 117 |
+
^^^^^^^^^^^^^^
|
| 118 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 119 |
+
return self._call_impl(*args, **kwargs)
|
| 120 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 121 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 122 |
+
return forward_call(*args, **kwargs)
|
| 123 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 124 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 125 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 126 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 127 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
|
| 128 |
+
return super().__call__(*args, **kwargs)
|
| 129 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 130 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 131 |
+
return self._call_impl(*args, **kwargs)
|
| 132 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 133 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 134 |
+
return inner()
|
| 135 |
+
^^^^^^^
|
| 136 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 137 |
+
result = forward_call(*args, **kwargs)
|
| 138 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 139 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
|
| 140 |
+
hidden_states = self.mlp(hidden_states)
|
| 141 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 142 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 143 |
+
return self._call_impl(*args, **kwargs)
|
| 144 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 145 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 146 |
+
return forward_call(*args, **kwargs)
|
| 147 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 148 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
|
| 149 |
+
down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 150 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 151 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 152 |
+
return self._call_impl(*args, **kwargs)
|
| 153 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 154 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 155 |
+
return forward_call(*args, **kwargs)
|
| 156 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 157 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/activations.py", line 103, in forward
|
| 158 |
+
return nn.functional.silu(input)
|
| 159 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 160 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/functional.py", line 2397, in silu
|
| 161 |
+
return torch._C._nn.silu(input)
|
| 162 |
+
^^^^^^^^^^^^^^^^^^^^^^^^
|
| 163 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 1.19 GiB. GPU 0 has a total capacity of 139.80 GiB of which 583.06 MiB is free. Including non-PyTorch memory, this process has 139.22 GiB memory in use. Of the allocated memory 136.99 GiB is allocated by PyTorch, and 1.03 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 164 |
+
[rank0]: Traceback (most recent call last):
|
| 165 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 166 |
+
[rank0]: main()
|
| 167 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 168 |
+
[rank0]: outputs = parallel_model(**batch)
|
| 169 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 170 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 171 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 172 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 173 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 174 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 175 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 176 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 177 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 178 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 179 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 180 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 181 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 182 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 183 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 184 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 185 |
+
[rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
|
| 186 |
+
[rank0]: outputs = self.base_causallm(
|
| 187 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^
|
| 188 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 189 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 190 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 191 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 192 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 193 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 194 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
|
| 195 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 196 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 197 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
|
| 198 |
+
[rank0]: outputs: BaseModelOutputWithPast = self.model(
|
| 199 |
+
[rank0]: ^^^^^^^^^^^
|
| 200 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 201 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 202 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 203 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 204 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 205 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 206 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
|
| 207 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 208 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 209 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
|
| 210 |
+
[rank0]: outputs = func(self, *args, **kwargs)
|
| 211 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 212 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
|
| 213 |
+
[rank0]: hidden_states = decoder_layer(
|
| 214 |
+
[rank0]: ^^^^^^^^^^^^^^
|
| 215 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 216 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 217 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 218 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 219 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 220 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 221 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 222 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 223 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 224 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
|
| 225 |
+
[rank0]: return super().__call__(*args, **kwargs)
|
| 226 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 227 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 228 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 229 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 230 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 231 |
+
[rank0]: return inner()
|
| 232 |
+
[rank0]: ^^^^^^^
|
| 233 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 234 |
+
[rank0]: result = forward_call(*args, **kwargs)
|
| 235 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 236 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
|
| 237 |
+
[rank0]: hidden_states = self.mlp(hidden_states)
|
| 238 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 239 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 240 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 241 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 242 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 243 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 244 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 245 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
|
| 246 |
+
[rank0]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 247 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 248 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 249 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 250 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 251 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 252 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 253 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 254 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/activations.py", line 103, in forward
|
| 255 |
+
[rank0]: return nn.functional.silu(input)
|
| 256 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 257 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/functional.py", line 2397, in silu
|
| 258 |
+
[rank0]: return torch._C._nn.silu(input)
|
| 259 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^
|
| 260 |
+
[rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 1.19 GiB. GPU 0 has a total capacity of 139.80 GiB of which 583.06 MiB is free. Including non-PyTorch memory, this process has 139.22 GiB memory in use. Of the allocated memory 136.99 GiB is allocated by PyTorch, and 1.03 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 261 |
+
[1;34mwandb[0m:
|
| 262 |
+
[1;34mwandb[0m: 🚀 View run [33mstar-sweep-bs128-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-46d2d3b7[0m at: [34mhttps://wandb.ai/seyedparsa/thoughtformer/runs/efpd3uen[0m
|
| 263 |
+
[1;34mwandb[0m: Find logs at: [1;35mwandb/run-20260530_033220-efpd3uen/logs[0m
|
| 264 |
+
E0530 03:32:31.760000 464943 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 465570) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
|
| 265 |
+
Traceback (most recent call last):
|
| 266 |
+
File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
|
| 267 |
+
sys.exit(main())
|
| 268 |
+
^^^^^^
|
| 269 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
|
| 270 |
+
return f(*args, **kwargs)
|
| 271 |
+
^^^^^^^^^^^^^^^^^^
|
| 272 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
|
| 273 |
+
run(args)
|
| 274 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
|
| 275 |
+
elastic_launch(
|
| 276 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
|
| 277 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 278 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 279 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
|
| 280 |
+
raise ChildFailedError(
|
| 281 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 282 |
+
============================================================
|
| 283 |
+
run.py FAILED
|
| 284 |
+
------------------------------------------------------------
|
| 285 |
+
Failures:
|
| 286 |
+
<NO_OTHER_FAILURES>
|
| 287 |
+
------------------------------------------------------------
|
| 288 |
+
Root Cause (first observed failure):
|
| 289 |
+
[0]:
|
| 290 |
+
time : 2026-05-30_03:32:31
|
| 291 |
+
host : ip-172-31-10-226.us-east-2.compute.internal
|
| 292 |
+
rank : 0 (local_rank: 0)
|
| 293 |
+
exitcode : 1 (pid: 465570)
|
| 294 |
+
error_file: <N/A>
|
| 295 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 296 |
+
============================================================
|
logs_backup/star_sweep_bs256_lr1e-4.log
ADDED
|
@@ -0,0 +1,290 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs256-lr1e-4', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 256, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 0.0001, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
|
| 2 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 3 |
+
return func(*args, **kwargs)
|
| 4 |
+
[rank0]:[W530 03:31:59.134090576 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 5 |
+
|
| 6 |
+
Running FSDP on rank = 0, world size = 1
|
| 7 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
|
| 8 |
+
_init_core_state(
|
| 9 |
+
FullyShardedDataParallel(
|
| 10 |
+
(_fsdp_wrapped_module): Coconut(
|
| 11 |
+
(base_causallm): Qwen3ForCausalLM(
|
| 12 |
+
(model): Qwen3Model(
|
| 13 |
+
(embed_tokens): Embedding(151672, 1024)
|
| 14 |
+
(layers): ModuleList(
|
| 15 |
+
(0-27): 28 x FullyShardedDataParallel(
|
| 16 |
+
(_fsdp_wrapped_module): Qwen3DecoderLayer(
|
| 17 |
+
(self_attn): Qwen3Attention(
|
| 18 |
+
(q_proj): Linear(in_features=1024, out_features=2048, bias=False)
|
| 19 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 20 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 21 |
+
(o_proj): Linear(in_features=2048, out_features=1024, bias=False)
|
| 22 |
+
(q_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 23 |
+
(k_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 24 |
+
)
|
| 25 |
+
(mlp): Qwen3MLP(
|
| 26 |
+
(gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 27 |
+
(up_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 28 |
+
(down_proj): Linear(in_features=3072, out_features=1024, bias=False)
|
| 29 |
+
(act_fn): SiLUActivation()
|
| 30 |
+
)
|
| 31 |
+
(input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 32 |
+
(post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 33 |
+
)
|
| 34 |
+
)
|
| 35 |
+
)
|
| 36 |
+
(norm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 37 |
+
(rotary_emb): Qwen3RotaryEmbedding()
|
| 38 |
+
)
|
| 39 |
+
(lm_head): Linear(in_features=1024, out_features=151936, bias=False)
|
| 40 |
+
)
|
| 41 |
+
(embedding): Embedding(151672, 1024)
|
| 42 |
+
)
|
| 43 |
+
)
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
|
| 47 |
+
wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 48 |
+
wandb: Tracking run with wandb version 0.25.1
|
| 49 |
+
wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033226-wj48phx5
|
| 50 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 51 |
+
wandb: Syncing run star-sweep-bs256-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-ff592230
|
| 52 |
+
wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
|
| 53 |
+
wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/wj48phx5
|
| 54 |
+
|
| 55 |
+
============================================================
|
| 56 |
+
EPOCH 0/30 (thought_stage=1, data_stage=1)
|
| 57 |
+
============================================================
|
| 58 |
+
thought_stage=1, c_thought=1, max_difficulty=1
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
LR schedule: cosine, warmup=1 steps, total=1170 steps (max_steps/epoch=39)
|
| 65 |
+
|
| 66 |
+
wandb: WARNING Serializing object of type str that is 3353666 bytes
|
| 67 |
+
Traceback (most recent call last):
|
| 68 |
+
File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 69 |
+
main()
|
| 70 |
+
File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 71 |
+
outputs = parallel_model(**batch)
|
| 72 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 73 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 74 |
+
return self._call_impl(*args, **kwargs)
|
| 75 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 76 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 77 |
+
return forward_call(*args, **kwargs)
|
| 78 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 79 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 80 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 81 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 82 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 83 |
+
return self._call_impl(*args, **kwargs)
|
| 84 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 85 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 86 |
+
return forward_call(*args, **kwargs)
|
| 87 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 88 |
+
File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
|
| 89 |
+
outputs = self.base_causallm(
|
| 90 |
+
^^^^^^^^^^^^^^^^^^^
|
| 91 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 92 |
+
return self._call_impl(*args, **kwargs)
|
| 93 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 94 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 95 |
+
return forward_call(*args, **kwargs)
|
| 96 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 97 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
|
| 98 |
+
output = func(self, *args, **kwargs)
|
| 99 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 100 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
|
| 101 |
+
outputs: BaseModelOutputWithPast = self.model(
|
| 102 |
+
^^^^^^^^^^^
|
| 103 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 104 |
+
return self._call_impl(*args, **kwargs)
|
| 105 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 106 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 107 |
+
return forward_call(*args, **kwargs)
|
| 108 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 109 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
|
| 110 |
+
output = func(self, *args, **kwargs)
|
| 111 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 112 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
|
| 113 |
+
outputs = func(self, *args, **kwargs)
|
| 114 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 115 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
|
| 116 |
+
hidden_states = decoder_layer(
|
| 117 |
+
^^^^^^^^^^^^^^
|
| 118 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 119 |
+
return self._call_impl(*args, **kwargs)
|
| 120 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 121 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 122 |
+
return forward_call(*args, **kwargs)
|
| 123 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 124 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 125 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 126 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 127 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
|
| 128 |
+
return super().__call__(*args, **kwargs)
|
| 129 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 130 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 131 |
+
return self._call_impl(*args, **kwargs)
|
| 132 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 133 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 134 |
+
return inner()
|
| 135 |
+
^^^^^^^
|
| 136 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 137 |
+
result = forward_call(*args, **kwargs)
|
| 138 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 139 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
|
| 140 |
+
hidden_states = self.mlp(hidden_states)
|
| 141 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 142 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 143 |
+
return self._call_impl(*args, **kwargs)
|
| 144 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 145 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 146 |
+
return forward_call(*args, **kwargs)
|
| 147 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 148 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
|
| 149 |
+
down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 150 |
+
^^^^^^^^^^^^^^^
|
| 151 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 152 |
+
return self._call_impl(*args, **kwargs)
|
| 153 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 154 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 155 |
+
return forward_call(*args, **kwargs)
|
| 156 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 157 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/linear.py", line 134, in forward
|
| 158 |
+
return F.linear(input, self.weight, self.bias)
|
| 159 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 160 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 2.38 GiB. GPU 0 has a total capacity of 139.80 GiB of which 1.40 GiB is free. Including non-PyTorch memory, this process has 138.39 GiB memory in use. Of the allocated memory 135.99 GiB is allocated by PyTorch, and 1.20 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 161 |
+
[rank0]: Traceback (most recent call last):
|
| 162 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 163 |
+
[rank0]: main()
|
| 164 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 165 |
+
[rank0]: outputs = parallel_model(**batch)
|
| 166 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 167 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 168 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 169 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 170 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 171 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 172 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 173 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 174 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 175 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 176 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 177 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 178 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 179 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 180 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 181 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 182 |
+
[rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
|
| 183 |
+
[rank0]: outputs = self.base_causallm(
|
| 184 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^
|
| 185 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 186 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 187 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 188 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 189 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 190 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 191 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
|
| 192 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 193 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 194 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
|
| 195 |
+
[rank0]: outputs: BaseModelOutputWithPast = self.model(
|
| 196 |
+
[rank0]: ^^^^^^^^^^^
|
| 197 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 198 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 199 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 200 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 201 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 202 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 203 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
|
| 204 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 205 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 206 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
|
| 207 |
+
[rank0]: outputs = func(self, *args, **kwargs)
|
| 208 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 209 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
|
| 210 |
+
[rank0]: hidden_states = decoder_layer(
|
| 211 |
+
[rank0]: ^^^^^^^^^^^^^^
|
| 212 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 213 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 214 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 215 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 216 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 217 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 218 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 219 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 220 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 221 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
|
| 222 |
+
[rank0]: return super().__call__(*args, **kwargs)
|
| 223 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 224 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 225 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 226 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 227 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 228 |
+
[rank0]: return inner()
|
| 229 |
+
[rank0]: ^^^^^^^
|
| 230 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 231 |
+
[rank0]: result = forward_call(*args, **kwargs)
|
| 232 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 233 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
|
| 234 |
+
[rank0]: hidden_states = self.mlp(hidden_states)
|
| 235 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 236 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 237 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 238 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 239 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 240 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 241 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 242 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
|
| 243 |
+
[rank0]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 244 |
+
[rank0]: ^^^^^^^^^^^^^^^
|
| 245 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 246 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 247 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 248 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 249 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 250 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 251 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/linear.py", line 134, in forward
|
| 252 |
+
[rank0]: return F.linear(input, self.weight, self.bias)
|
| 253 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 254 |
+
[rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 2.38 GiB. GPU 0 has a total capacity of 139.80 GiB of which 1.40 GiB is free. Including non-PyTorch memory, this process has 138.39 GiB memory in use. Of the allocated memory 135.99 GiB is allocated by PyTorch, and 1.20 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 255 |
+
[1;34mwandb[0m:
|
| 256 |
+
[1;34mwandb[0m: 🚀 View run [33mstar-sweep-bs256-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-ff592230[0m at: [34mhttps://wandb.ai/seyedparsa/thoughtformer/runs/wj48phx5[0m
|
| 257 |
+
[1;34mwandb[0m: Find logs at: [1;35mwandb/run-20260530_033226-wj48phx5/logs[0m
|
| 258 |
+
E0530 03:32:40.146000 466881 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 467529) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
|
| 259 |
+
Traceback (most recent call last):
|
| 260 |
+
File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
|
| 261 |
+
sys.exit(main())
|
| 262 |
+
^^^^^^
|
| 263 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
|
| 264 |
+
return f(*args, **kwargs)
|
| 265 |
+
^^^^^^^^^^^^^^^^^^
|
| 266 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
|
| 267 |
+
run(args)
|
| 268 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
|
| 269 |
+
elastic_launch(
|
| 270 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
|
| 271 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 272 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 273 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
|
| 274 |
+
raise ChildFailedError(
|
| 275 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 276 |
+
============================================================
|
| 277 |
+
run.py FAILED
|
| 278 |
+
------------------------------------------------------------
|
| 279 |
+
Failures:
|
| 280 |
+
<NO_OTHER_FAILURES>
|
| 281 |
+
------------------------------------------------------------
|
| 282 |
+
Root Cause (first observed failure):
|
| 283 |
+
[0]:
|
| 284 |
+
time : 2026-05-30_03:32:40
|
| 285 |
+
host : ip-172-31-10-226.us-east-2.compute.internal
|
| 286 |
+
rank : 0 (local_rank: 0)
|
| 287 |
+
exitcode : 1 (pid: 467529)
|
| 288 |
+
error_file: <N/A>
|
| 289 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 290 |
+
============================================================
|
logs_backup/star_sweep_bs256_lr1e-5.log
ADDED
|
@@ -0,0 +1,290 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs256-lr1e-5', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 256, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 1e-05, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
|
| 2 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 3 |
+
return func(*args, **kwargs)
|
| 4 |
+
[rank0]:[W530 03:31:57.089584891 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 5 |
+
|
| 6 |
+
Running FSDP on rank = 0, world size = 1
|
| 7 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
|
| 8 |
+
_init_core_state(
|
| 9 |
+
FullyShardedDataParallel(
|
| 10 |
+
(_fsdp_wrapped_module): Coconut(
|
| 11 |
+
(base_causallm): Qwen3ForCausalLM(
|
| 12 |
+
(model): Qwen3Model(
|
| 13 |
+
(embed_tokens): Embedding(151672, 1024)
|
| 14 |
+
(layers): ModuleList(
|
| 15 |
+
(0-27): 28 x FullyShardedDataParallel(
|
| 16 |
+
(_fsdp_wrapped_module): Qwen3DecoderLayer(
|
| 17 |
+
(self_attn): Qwen3Attention(
|
| 18 |
+
(q_proj): Linear(in_features=1024, out_features=2048, bias=False)
|
| 19 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 20 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 21 |
+
(o_proj): Linear(in_features=2048, out_features=1024, bias=False)
|
| 22 |
+
(q_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 23 |
+
(k_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 24 |
+
)
|
| 25 |
+
(mlp): Qwen3MLP(
|
| 26 |
+
(gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 27 |
+
(up_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 28 |
+
(down_proj): Linear(in_features=3072, out_features=1024, bias=False)
|
| 29 |
+
(act_fn): SiLUActivation()
|
| 30 |
+
)
|
| 31 |
+
(input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 32 |
+
(post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 33 |
+
)
|
| 34 |
+
)
|
| 35 |
+
)
|
| 36 |
+
(norm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 37 |
+
(rotary_emb): Qwen3RotaryEmbedding()
|
| 38 |
+
)
|
| 39 |
+
(lm_head): Linear(in_features=1024, out_features=151936, bias=False)
|
| 40 |
+
)
|
| 41 |
+
(embedding): Embedding(151672, 1024)
|
| 42 |
+
)
|
| 43 |
+
)
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
|
| 47 |
+
wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 48 |
+
wandb: Tracking run with wandb version 0.25.1
|
| 49 |
+
wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033224-tjrts1mq
|
| 50 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 51 |
+
wandb: Syncing run star-sweep-bs256-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-a9a8c070
|
| 52 |
+
wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
|
| 53 |
+
wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/tjrts1mq
|
| 54 |
+
|
| 55 |
+
============================================================
|
| 56 |
+
EPOCH 0/30 (thought_stage=1, data_stage=1)
|
| 57 |
+
============================================================
|
| 58 |
+
thought_stage=1, c_thought=1, max_difficulty=1
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
LR schedule: cosine, warmup=1 steps, total=1170 steps (max_steps/epoch=39)
|
| 65 |
+
|
| 66 |
+
wandb: WARNING Serializing object of type str that is 3353666 bytes
|
| 67 |
+
Traceback (most recent call last):
|
| 68 |
+
File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 69 |
+
main()
|
| 70 |
+
File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 71 |
+
outputs = parallel_model(**batch)
|
| 72 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 73 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 74 |
+
return self._call_impl(*args, **kwargs)
|
| 75 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 76 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 77 |
+
return forward_call(*args, **kwargs)
|
| 78 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 79 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 80 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 81 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 82 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 83 |
+
return self._call_impl(*args, **kwargs)
|
| 84 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 85 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 86 |
+
return forward_call(*args, **kwargs)
|
| 87 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 88 |
+
File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
|
| 89 |
+
outputs = self.base_causallm(
|
| 90 |
+
^^^^^^^^^^^^^^^^^^^
|
| 91 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 92 |
+
return self._call_impl(*args, **kwargs)
|
| 93 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 94 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 95 |
+
return forward_call(*args, **kwargs)
|
| 96 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 97 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
|
| 98 |
+
output = func(self, *args, **kwargs)
|
| 99 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 100 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
|
| 101 |
+
outputs: BaseModelOutputWithPast = self.model(
|
| 102 |
+
^^^^^^^^^^^
|
| 103 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 104 |
+
return self._call_impl(*args, **kwargs)
|
| 105 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 106 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 107 |
+
return forward_call(*args, **kwargs)
|
| 108 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 109 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
|
| 110 |
+
output = func(self, *args, **kwargs)
|
| 111 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 112 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
|
| 113 |
+
outputs = func(self, *args, **kwargs)
|
| 114 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 115 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
|
| 116 |
+
hidden_states = decoder_layer(
|
| 117 |
+
^^^^^^^^^^^^^^
|
| 118 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 119 |
+
return self._call_impl(*args, **kwargs)
|
| 120 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 121 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 122 |
+
return forward_call(*args, **kwargs)
|
| 123 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 124 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 125 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 126 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 127 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
|
| 128 |
+
return super().__call__(*args, **kwargs)
|
| 129 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 130 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 131 |
+
return self._call_impl(*args, **kwargs)
|
| 132 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 133 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 134 |
+
return inner()
|
| 135 |
+
^^^^^^^
|
| 136 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 137 |
+
result = forward_call(*args, **kwargs)
|
| 138 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 139 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
|
| 140 |
+
hidden_states = self.mlp(hidden_states)
|
| 141 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 142 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 143 |
+
return self._call_impl(*args, **kwargs)
|
| 144 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 145 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 146 |
+
return forward_call(*args, **kwargs)
|
| 147 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 148 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
|
| 149 |
+
down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 150 |
+
^^^^^^^^^^^^^^^
|
| 151 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 152 |
+
return self._call_impl(*args, **kwargs)
|
| 153 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 154 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 155 |
+
return forward_call(*args, **kwargs)
|
| 156 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 157 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/linear.py", line 134, in forward
|
| 158 |
+
return F.linear(input, self.weight, self.bias)
|
| 159 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 160 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 2.38 GiB. GPU 0 has a total capacity of 139.80 GiB of which 1.40 GiB is free. Including non-PyTorch memory, this process has 138.39 GiB memory in use. Of the allocated memory 135.99 GiB is allocated by PyTorch, and 1.20 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 161 |
+
[rank0]: Traceback (most recent call last):
|
| 162 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 163 |
+
[rank0]: main()
|
| 164 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 165 |
+
[rank0]: outputs = parallel_model(**batch)
|
| 166 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 167 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 168 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 169 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 170 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 171 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 172 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 173 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 174 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 175 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 176 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 177 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 178 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 179 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 180 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 181 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 182 |
+
[rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
|
| 183 |
+
[rank0]: outputs = self.base_causallm(
|
| 184 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^
|
| 185 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 186 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 187 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 188 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 189 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 190 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 191 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
|
| 192 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 193 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 194 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
|
| 195 |
+
[rank0]: outputs: BaseModelOutputWithPast = self.model(
|
| 196 |
+
[rank0]: ^^^^^^^^^^^
|
| 197 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 198 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 199 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 200 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 201 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 202 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 203 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
|
| 204 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 205 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 206 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
|
| 207 |
+
[rank0]: outputs = func(self, *args, **kwargs)
|
| 208 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 209 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
|
| 210 |
+
[rank0]: hidden_states = decoder_layer(
|
| 211 |
+
[rank0]: ^^^^^^^^^^^^^^
|
| 212 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 213 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 214 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 215 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 216 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 217 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 218 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 219 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 220 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 221 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
|
| 222 |
+
[rank0]: return super().__call__(*args, **kwargs)
|
| 223 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 224 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 225 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 226 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 227 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 228 |
+
[rank0]: return inner()
|
| 229 |
+
[rank0]: ^^^^^^^
|
| 230 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 231 |
+
[rank0]: result = forward_call(*args, **kwargs)
|
| 232 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 233 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
|
| 234 |
+
[rank0]: hidden_states = self.mlp(hidden_states)
|
| 235 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 236 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 237 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 238 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 239 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 240 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 241 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 242 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
|
| 243 |
+
[rank0]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 244 |
+
[rank0]: ^^^^^^^^^^^^^^^
|
| 245 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 246 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 247 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 248 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 249 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 250 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 251 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/linear.py", line 134, in forward
|
| 252 |
+
[rank0]: return F.linear(input, self.weight, self.bias)
|
| 253 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 254 |
+
[rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 2.38 GiB. GPU 0 has a total capacity of 139.80 GiB of which 1.40 GiB is free. Including non-PyTorch memory, this process has 138.39 GiB memory in use. Of the allocated memory 135.99 GiB is allocated by PyTorch, and 1.20 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 255 |
+
[1;34mwandb[0m:
|
| 256 |
+
[1;34mwandb[0m: 🚀 View run [33mstar-sweep-bs256-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-a9a8c070[0m at: [34mhttps://wandb.ai/seyedparsa/thoughtformer/runs/tjrts1mq[0m
|
| 257 |
+
[1;34mwandb[0m: Find logs at: [1;35mwandb/run-20260530_033224-tjrts1mq/logs[0m
|
| 258 |
+
E0530 03:32:37.865000 466226 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 466874) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
|
| 259 |
+
Traceback (most recent call last):
|
| 260 |
+
File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
|
| 261 |
+
sys.exit(main())
|
| 262 |
+
^^^^^^
|
| 263 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
|
| 264 |
+
return f(*args, **kwargs)
|
| 265 |
+
^^^^^^^^^^^^^^^^^^
|
| 266 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
|
| 267 |
+
run(args)
|
| 268 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
|
| 269 |
+
elastic_launch(
|
| 270 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
|
| 271 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 272 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 273 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
|
| 274 |
+
raise ChildFailedError(
|
| 275 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 276 |
+
============================================================
|
| 277 |
+
run.py FAILED
|
| 278 |
+
------------------------------------------------------------
|
| 279 |
+
Failures:
|
| 280 |
+
<NO_OTHER_FAILURES>
|
| 281 |
+
------------------------------------------------------------
|
| 282 |
+
Root Cause (first observed failure):
|
| 283 |
+
[0]:
|
| 284 |
+
time : 2026-05-30_03:32:37
|
| 285 |
+
host : ip-172-31-10-226.us-east-2.compute.internal
|
| 286 |
+
rank : 0 (local_rank: 0)
|
| 287 |
+
exitcode : 1 (pid: 466874)
|
| 288 |
+
error_file: <N/A>
|
| 289 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 290 |
+
============================================================
|
logs_backup/star_sweep_bs32_lr1e-4.log
ADDED
|
@@ -0,0 +1,152 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs32-lr1e-4', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 32, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 0.0001, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
|
| 2 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 3 |
+
return func(*args, **kwargs)
|
| 4 |
+
[rank0]:[W530 03:31:47.851491967 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 5 |
+
|
| 6 |
+
Running FSDP on rank = 0, world size = 1
|
| 7 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
|
| 8 |
+
_init_core_state(
|
| 9 |
+
FullyShardedDataParallel(
|
| 10 |
+
(_fsdp_wrapped_module): Coconut(
|
| 11 |
+
(base_causallm): Qwen3ForCausalLM(
|
| 12 |
+
(model): Qwen3Model(
|
| 13 |
+
(embed_tokens): Embedding(151672, 1024)
|
| 14 |
+
(layers): ModuleList(
|
| 15 |
+
(0-27): 28 x FullyShardedDataParallel(
|
| 16 |
+
(_fsdp_wrapped_module): Qwen3DecoderLayer(
|
| 17 |
+
(self_attn): Qwen3Attention(
|
| 18 |
+
(q_proj): Linear(in_features=1024, out_features=2048, bias=False)
|
| 19 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 20 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 21 |
+
(o_proj): Linear(in_features=2048, out_features=1024, bias=False)
|
| 22 |
+
(q_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 23 |
+
(k_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 24 |
+
)
|
| 25 |
+
(mlp): Qwen3MLP(
|
| 26 |
+
(gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 27 |
+
(up_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 28 |
+
(down_proj): Linear(in_features=3072, out_features=1024, bias=False)
|
| 29 |
+
(act_fn): SiLUActivation()
|
| 30 |
+
)
|
| 31 |
+
(input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 32 |
+
(post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 33 |
+
)
|
| 34 |
+
)
|
| 35 |
+
)
|
| 36 |
+
(norm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 37 |
+
(rotary_emb): Qwen3RotaryEmbedding()
|
| 38 |
+
)
|
| 39 |
+
(lm_head): Linear(in_features=1024, out_features=151936, bias=False)
|
| 40 |
+
)
|
| 41 |
+
(embedding): Embedding(151672, 1024)
|
| 42 |
+
)
|
| 43 |
+
)
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
|
| 47 |
+
wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 48 |
+
wandb: Tracking run with wandb version 0.25.1
|
| 49 |
+
wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033214-gu4dqvfl
|
| 50 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 51 |
+
wandb: Syncing run star-sweep-bs32-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-d2591099
|
| 52 |
+
wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
|
| 53 |
+
wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/gu4dqvfl
|
| 54 |
+
|
| 55 |
+
============================================================
|
| 56 |
+
EPOCH 0/30 (thought_stage=1, data_stage=1)
|
| 57 |
+
============================================================
|
| 58 |
+
thought_stage=1, c_thought=1, max_difficulty=1
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
LR schedule: cosine, warmup=15 steps, total=9360 steps (max_steps/epoch=312)
|
| 65 |
+
|
| 66 |
+
wandb: WARNING Serializing object of type str that is 419185 bytes
|
| 67 |
+
Traceback (most recent call last):
|
| 68 |
+
File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 69 |
+
main()
|
| 70 |
+
File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 71 |
+
outputs = parallel_model(**batch)
|
| 72 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 73 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 74 |
+
return self._call_impl(*args, **kwargs)
|
| 75 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 76 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 77 |
+
return forward_call(*args, **kwargs)
|
| 78 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 79 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 80 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 81 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 82 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 83 |
+
return self._call_impl(*args, **kwargs)
|
| 84 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 85 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 86 |
+
return forward_call(*args, **kwargs)
|
| 87 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 88 |
+
File "/home/ubuntu/thoughtformer/coconut.py", line 239, in forward
|
| 89 |
+
logits = torch.cat(logits, dim=-2)
|
| 90 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 91 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 14.77 GiB. GPU 0 has a total capacity of 139.80 GiB of which 6.10 GiB is free. Including non-PyTorch memory, this process has 133.69 GiB memory in use. Of the allocated memory 131.74 GiB is allocated by PyTorch, and 767.15 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 92 |
+
[rank0]: Traceback (most recent call last):
|
| 93 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 94 |
+
[rank0]: main()
|
| 95 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 96 |
+
[rank0]: outputs = parallel_model(**batch)
|
| 97 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 98 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 99 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 100 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 101 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 102 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 103 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 104 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 105 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 106 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 107 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 108 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 109 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 110 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 111 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 112 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 113 |
+
[rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 239, in forward
|
| 114 |
+
[rank0]: logits = torch.cat(logits, dim=-2)
|
| 115 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 116 |
+
[rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 14.77 GiB. GPU 0 has a total capacity of 139.80 GiB of which 6.10 GiB is free. Including non-PyTorch memory, this process has 133.69 GiB memory in use. Of the allocated memory 131.74 GiB is allocated by PyTorch, and 767.15 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 117 |
+
[1;34mwandb[0m:
|
| 118 |
+
[1;34mwandb[0m: 🚀 View run [33mstar-sweep-bs32-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-d2591099[0m at: [34mhttps://wandb.ai/seyedparsa/thoughtformer/runs/gu4dqvfl[0m
|
| 119 |
+
[1;34mwandb[0m: Find logs at: [1;35mwandb/run-20260530_033214-gu4dqvfl/logs[0m
|
| 120 |
+
E0530 03:32:24.440000 464181 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 464254) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
|
| 121 |
+
Traceback (most recent call last):
|
| 122 |
+
File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
|
| 123 |
+
sys.exit(main())
|
| 124 |
+
^^^^^^
|
| 125 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
|
| 126 |
+
return f(*args, **kwargs)
|
| 127 |
+
^^^^^^^^^^^^^^^^^^
|
| 128 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
|
| 129 |
+
run(args)
|
| 130 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
|
| 131 |
+
elastic_launch(
|
| 132 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
|
| 133 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 134 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 135 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
|
| 136 |
+
raise ChildFailedError(
|
| 137 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 138 |
+
============================================================
|
| 139 |
+
run.py FAILED
|
| 140 |
+
------------------------------------------------------------
|
| 141 |
+
Failures:
|
| 142 |
+
<NO_OTHER_FAILURES>
|
| 143 |
+
------------------------------------------------------------
|
| 144 |
+
Root Cause (first observed failure):
|
| 145 |
+
[0]:
|
| 146 |
+
time : 2026-05-30_03:32:24
|
| 147 |
+
host : ip-172-31-10-226.us-east-2.compute.internal
|
| 148 |
+
rank : 0 (local_rank: 0)
|
| 149 |
+
exitcode : 1 (pid: 464254)
|
| 150 |
+
error_file: <N/A>
|
| 151 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 152 |
+
============================================================
|
logs_backup/star_sweep_bs32_lr1e-5.log
ADDED
|
@@ -0,0 +1,152 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs32-lr1e-5', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 32, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 1e-05, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
|
| 2 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 3 |
+
return func(*args, **kwargs)
|
| 4 |
+
[rank0]:[W530 03:31:45.823267663 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 5 |
+
|
| 6 |
+
Running FSDP on rank = 0, world size = 1
|
| 7 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
|
| 8 |
+
_init_core_state(
|
| 9 |
+
FullyShardedDataParallel(
|
| 10 |
+
(_fsdp_wrapped_module): Coconut(
|
| 11 |
+
(base_causallm): Qwen3ForCausalLM(
|
| 12 |
+
(model): Qwen3Model(
|
| 13 |
+
(embed_tokens): Embedding(151672, 1024)
|
| 14 |
+
(layers): ModuleList(
|
| 15 |
+
(0-27): 28 x FullyShardedDataParallel(
|
| 16 |
+
(_fsdp_wrapped_module): Qwen3DecoderLayer(
|
| 17 |
+
(self_attn): Qwen3Attention(
|
| 18 |
+
(q_proj): Linear(in_features=1024, out_features=2048, bias=False)
|
| 19 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 20 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 21 |
+
(o_proj): Linear(in_features=2048, out_features=1024, bias=False)
|
| 22 |
+
(q_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 23 |
+
(k_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 24 |
+
)
|
| 25 |
+
(mlp): Qwen3MLP(
|
| 26 |
+
(gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 27 |
+
(up_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 28 |
+
(down_proj): Linear(in_features=3072, out_features=1024, bias=False)
|
| 29 |
+
(act_fn): SiLUActivation()
|
| 30 |
+
)
|
| 31 |
+
(input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 32 |
+
(post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 33 |
+
)
|
| 34 |
+
)
|
| 35 |
+
)
|
| 36 |
+
(norm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 37 |
+
(rotary_emb): Qwen3RotaryEmbedding()
|
| 38 |
+
)
|
| 39 |
+
(lm_head): Linear(in_features=1024, out_features=151936, bias=False)
|
| 40 |
+
)
|
| 41 |
+
(embedding): Embedding(151672, 1024)
|
| 42 |
+
)
|
| 43 |
+
)
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
|
| 47 |
+
wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 48 |
+
wandb: Tracking run with wandb version 0.25.1
|
| 49 |
+
wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033212-jek5luxr
|
| 50 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 51 |
+
wandb: Syncing run star-sweep-bs32-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-c3e91928
|
| 52 |
+
wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
|
| 53 |
+
wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/jek5luxr
|
| 54 |
+
|
| 55 |
+
============================================================
|
| 56 |
+
EPOCH 0/30 (thought_stage=1, data_stage=1)
|
| 57 |
+
============================================================
|
| 58 |
+
thought_stage=1, c_thought=1, max_difficulty=1
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
LR schedule: cosine, warmup=15 steps, total=9360 steps (max_steps/epoch=312)
|
| 65 |
+
|
| 66 |
+
wandb: WARNING Serializing object of type str that is 419185 bytes
|
| 67 |
+
Traceback (most recent call last):
|
| 68 |
+
File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 69 |
+
main()
|
| 70 |
+
File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 71 |
+
outputs = parallel_model(**batch)
|
| 72 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 73 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 74 |
+
return self._call_impl(*args, **kwargs)
|
| 75 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 76 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 77 |
+
return forward_call(*args, **kwargs)
|
| 78 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 79 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 80 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 81 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 82 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 83 |
+
return self._call_impl(*args, **kwargs)
|
| 84 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 85 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 86 |
+
return forward_call(*args, **kwargs)
|
| 87 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 88 |
+
File "/home/ubuntu/thoughtformer/coconut.py", line 239, in forward
|
| 89 |
+
logits = torch.cat(logits, dim=-2)
|
| 90 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 91 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 14.77 GiB. GPU 0 has a total capacity of 139.80 GiB of which 6.10 GiB is free. Including non-PyTorch memory, this process has 133.69 GiB memory in use. Of the allocated memory 131.74 GiB is allocated by PyTorch, and 767.15 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 92 |
+
[rank0]: Traceback (most recent call last):
|
| 93 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 94 |
+
[rank0]: main()
|
| 95 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 96 |
+
[rank0]: outputs = parallel_model(**batch)
|
| 97 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 98 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 99 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 100 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 101 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 102 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 103 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 104 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 105 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 106 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 107 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 108 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 109 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 110 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 111 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 112 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 113 |
+
[rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 239, in forward
|
| 114 |
+
[rank0]: logits = torch.cat(logits, dim=-2)
|
| 115 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 116 |
+
[rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 14.77 GiB. GPU 0 has a total capacity of 139.80 GiB of which 6.10 GiB is free. Including non-PyTorch memory, this process has 133.69 GiB memory in use. Of the allocated memory 131.74 GiB is allocated by PyTorch, and 767.15 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 117 |
+
[1;34mwandb[0m:
|
| 118 |
+
[1;34mwandb[0m: 🚀 View run [33mstar-sweep-bs32-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-c3e91928[0m at: [34mhttps://wandb.ai/seyedparsa/thoughtformer/runs/jek5luxr[0m
|
| 119 |
+
[1;34mwandb[0m: Find logs at: [1;35mwandb/run-20260530_033212-jek5luxr/logs[0m
|
| 120 |
+
E0530 03:32:23.853000 464041 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 464115) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
|
| 121 |
+
Traceback (most recent call last):
|
| 122 |
+
File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
|
| 123 |
+
sys.exit(main())
|
| 124 |
+
^^^^^^
|
| 125 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
|
| 126 |
+
return f(*args, **kwargs)
|
| 127 |
+
^^^^^^^^^^^^^^^^^^
|
| 128 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
|
| 129 |
+
run(args)
|
| 130 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
|
| 131 |
+
elastic_launch(
|
| 132 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
|
| 133 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 134 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 135 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
|
| 136 |
+
raise ChildFailedError(
|
| 137 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 138 |
+
============================================================
|
| 139 |
+
run.py FAILED
|
| 140 |
+
------------------------------------------------------------
|
| 141 |
+
Failures:
|
| 142 |
+
<NO_OTHER_FAILURES>
|
| 143 |
+
------------------------------------------------------------
|
| 144 |
+
Root Cause (first observed failure):
|
| 145 |
+
[0]:
|
| 146 |
+
time : 2026-05-30_03:32:23
|
| 147 |
+
host : ip-172-31-10-226.us-east-2.compute.internal
|
| 148 |
+
rank : 0 (local_rank: 0)
|
| 149 |
+
exitcode : 1 (pid: 464115)
|
| 150 |
+
error_file: <N/A>
|
| 151 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 152 |
+
============================================================
|
logs_backup/star_sweep_bs64_lr1e-4.log
ADDED
|
@@ -0,0 +1,285 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs64-lr1e-4', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 64, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 0.0001, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
|
| 2 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 3 |
+
return func(*args, **kwargs)
|
| 4 |
+
[rank0]:[W530 03:31:51.090585317 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 5 |
+
|
| 6 |
+
Running FSDP on rank = 0, world size = 1
|
| 7 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
|
| 8 |
+
_init_core_state(
|
| 9 |
+
FullyShardedDataParallel(
|
| 10 |
+
(_fsdp_wrapped_module): Coconut(
|
| 11 |
+
(base_causallm): Qwen3ForCausalLM(
|
| 12 |
+
(model): Qwen3Model(
|
| 13 |
+
(embed_tokens): Embedding(151672, 1024)
|
| 14 |
+
(layers): ModuleList(
|
| 15 |
+
(0-27): 28 x FullyShardedDataParallel(
|
| 16 |
+
(_fsdp_wrapped_module): Qwen3DecoderLayer(
|
| 17 |
+
(self_attn): Qwen3Attention(
|
| 18 |
+
(q_proj): Linear(in_features=1024, out_features=2048, bias=False)
|
| 19 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 20 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 21 |
+
(o_proj): Linear(in_features=2048, out_features=1024, bias=False)
|
| 22 |
+
(q_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 23 |
+
(k_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 24 |
+
)
|
| 25 |
+
(mlp): Qwen3MLP(
|
| 26 |
+
(gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 27 |
+
(up_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 28 |
+
(down_proj): Linear(in_features=3072, out_features=1024, bias=False)
|
| 29 |
+
(act_fn): SiLUActivation()
|
| 30 |
+
)
|
| 31 |
+
(input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 32 |
+
(post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 33 |
+
)
|
| 34 |
+
)
|
| 35 |
+
)
|
| 36 |
+
(norm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 37 |
+
(rotary_emb): Qwen3RotaryEmbedding()
|
| 38 |
+
)
|
| 39 |
+
(lm_head): Linear(in_features=1024, out_features=151936, bias=False)
|
| 40 |
+
)
|
| 41 |
+
(embedding): Embedding(151672, 1024)
|
| 42 |
+
)
|
| 43 |
+
)
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
|
| 47 |
+
wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 48 |
+
wandb: setting up run edlb2i8x
|
| 49 |
+
wandb: Tracking run with wandb version 0.25.1
|
| 50 |
+
wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033218-edlb2i8x
|
| 51 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 52 |
+
wandb: Syncing run star-sweep-bs64-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-64cda263
|
| 53 |
+
wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
|
| 54 |
+
wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/edlb2i8x
|
| 55 |
+
|
| 56 |
+
============================================================
|
| 57 |
+
EPOCH 0/30 (thought_stage=1, data_stage=1)
|
| 58 |
+
============================================================
|
| 59 |
+
thought_stage=1, c_thought=1, max_difficulty=1
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
|
| 65 |
+
LR schedule: cosine, warmup=7 steps, total=4680 steps (max_steps/epoch=156)
|
| 66 |
+
|
| 67 |
+
wandb: WARNING Serializing object of type str that is 838345 bytes
|
| 68 |
+
Traceback (most recent call last):
|
| 69 |
+
File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 70 |
+
main()
|
| 71 |
+
File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 72 |
+
outputs = parallel_model(**batch)
|
| 73 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 74 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 75 |
+
return self._call_impl(*args, **kwargs)
|
| 76 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 77 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 78 |
+
return forward_call(*args, **kwargs)
|
| 79 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 80 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 81 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 82 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 83 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 84 |
+
return self._call_impl(*args, **kwargs)
|
| 85 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 86 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 87 |
+
return forward_call(*args, **kwargs)
|
| 88 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 89 |
+
File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
|
| 90 |
+
outputs = self.base_causallm(
|
| 91 |
+
^^^^^^^^^^^^^^^^^^^
|
| 92 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 93 |
+
return self._call_impl(*args, **kwargs)
|
| 94 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 95 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 96 |
+
return forward_call(*args, **kwargs)
|
| 97 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 98 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
|
| 99 |
+
output = func(self, *args, **kwargs)
|
| 100 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 101 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
|
| 102 |
+
outputs: BaseModelOutputWithPast = self.model(
|
| 103 |
+
^^^^^^^^^^^
|
| 104 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 105 |
+
return self._call_impl(*args, **kwargs)
|
| 106 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 107 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 108 |
+
return forward_call(*args, **kwargs)
|
| 109 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 110 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
|
| 111 |
+
output = func(self, *args, **kwargs)
|
| 112 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 113 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
|
| 114 |
+
outputs = func(self, *args, **kwargs)
|
| 115 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 116 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
|
| 117 |
+
hidden_states = decoder_layer(
|
| 118 |
+
^^^^^^^^^^^^^^
|
| 119 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 120 |
+
return self._call_impl(*args, **kwargs)
|
| 121 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 122 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 123 |
+
return forward_call(*args, **kwargs)
|
| 124 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 125 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 126 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 127 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 128 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
|
| 129 |
+
return super().__call__(*args, **kwargs)
|
| 130 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 131 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 132 |
+
return self._call_impl(*args, **kwargs)
|
| 133 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 134 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 135 |
+
return inner()
|
| 136 |
+
^^^^^^^
|
| 137 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 138 |
+
result = forward_call(*args, **kwargs)
|
| 139 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 140 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 318, in forward
|
| 141 |
+
hidden_states, _ = self.self_attn(
|
| 142 |
+
^^^^^^^^^^^^^^^
|
| 143 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 144 |
+
return self._call_impl(*args, **kwargs)
|
| 145 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 146 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 147 |
+
return inner()
|
| 148 |
+
^^^^^^^
|
| 149 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 150 |
+
result = forward_call(*args, **kwargs)
|
| 151 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 152 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 277, in forward
|
| 153 |
+
attn_output, attn_weights = attention_interface(
|
| 154 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 155 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/integrations/sdpa_attention.py", line 92, in sdpa_attention_forward
|
| 156 |
+
attn_output = torch.nn.functional.scaled_dot_product_attention(
|
| 157 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 158 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 406.00 MiB. GPU 0 has a total capacity of 139.80 GiB of which 45.06 MiB is free. Including non-PyTorch memory, this process has 139.75 GiB memory in use. Of the allocated memory 137.33 GiB is allocated by PyTorch, and 1.22 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 159 |
+
[rank0]: Traceback (most recent call last):
|
| 160 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 161 |
+
[rank0]: main()
|
| 162 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 163 |
+
[rank0]: outputs = parallel_model(**batch)
|
| 164 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 165 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 166 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 167 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 168 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 169 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 170 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 171 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 172 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 173 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 174 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 175 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 176 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 177 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 178 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 179 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 180 |
+
[rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
|
| 181 |
+
[rank0]: outputs = self.base_causallm(
|
| 182 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^
|
| 183 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 184 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 185 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 186 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 187 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 188 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 189 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
|
| 190 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 191 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 192 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
|
| 193 |
+
[rank0]: outputs: BaseModelOutputWithPast = self.model(
|
| 194 |
+
[rank0]: ^^^^^^^^^^^
|
| 195 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 196 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 197 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 198 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 199 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 200 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 201 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
|
| 202 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 203 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 204 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
|
| 205 |
+
[rank0]: outputs = func(self, *args, **kwargs)
|
| 206 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 207 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
|
| 208 |
+
[rank0]: hidden_states = decoder_layer(
|
| 209 |
+
[rank0]: ^^^^^^^^^^^^^^
|
| 210 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 211 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 212 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 213 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 214 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 215 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 216 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 217 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 218 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 219 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
|
| 220 |
+
[rank0]: return super().__call__(*args, **kwargs)
|
| 221 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 222 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 223 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 224 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 225 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 226 |
+
[rank0]: return inner()
|
| 227 |
+
[rank0]: ^^^^^^^
|
| 228 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 229 |
+
[rank0]: result = forward_call(*args, **kwargs)
|
| 230 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 231 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 318, in forward
|
| 232 |
+
[rank0]: hidden_states, _ = self.self_attn(
|
| 233 |
+
[rank0]: ^^^^^^^^^^^^^^^
|
| 234 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 235 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 236 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 237 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 238 |
+
[rank0]: return inner()
|
| 239 |
+
[rank0]: ^^^^^^^
|
| 240 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 241 |
+
[rank0]: result = forward_call(*args, **kwargs)
|
| 242 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 243 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 277, in forward
|
| 244 |
+
[rank0]: attn_output, attn_weights = attention_interface(
|
| 245 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^
|
| 246 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/integrations/sdpa_attention.py", line 92, in sdpa_attention_forward
|
| 247 |
+
[rank0]: attn_output = torch.nn.functional.scaled_dot_product_attention(
|
| 248 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 249 |
+
[rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 406.00 MiB. GPU 0 has a total capacity of 139.80 GiB of which 45.06 MiB is free. Including non-PyTorch memory, this process has 139.75 GiB memory in use. Of the allocated memory 137.33 GiB is allocated by PyTorch, and 1.22 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 250 |
+
[1;34mwandb[0m:
|
| 251 |
+
[1;34mwandb[0m: 🚀 View run [33mstar-sweep-bs64-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-64cda263[0m at: [34mhttps://wandb.ai/seyedparsa/thoughtformer/runs/edlb2i8x[0m
|
| 252 |
+
[1;34mwandb[0m: Find logs at: [1;35mwandb/run-20260530_033218-edlb2i8x/logs[0m
|
| 253 |
+
E0530 03:32:29.958000 464466 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 464939) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
|
| 254 |
+
Traceback (most recent call last):
|
| 255 |
+
File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
|
| 256 |
+
sys.exit(main())
|
| 257 |
+
^^^^^^
|
| 258 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
|
| 259 |
+
return f(*args, **kwargs)
|
| 260 |
+
^^^^^^^^^^^^^^^^^^
|
| 261 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
|
| 262 |
+
run(args)
|
| 263 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
|
| 264 |
+
elastic_launch(
|
| 265 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
|
| 266 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 267 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 268 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
|
| 269 |
+
raise ChildFailedError(
|
| 270 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 271 |
+
============================================================
|
| 272 |
+
run.py FAILED
|
| 273 |
+
------------------------------------------------------------
|
| 274 |
+
Failures:
|
| 275 |
+
<NO_OTHER_FAILURES>
|
| 276 |
+
------------------------------------------------------------
|
| 277 |
+
Root Cause (first observed failure):
|
| 278 |
+
[0]:
|
| 279 |
+
time : 2026-05-30_03:32:29
|
| 280 |
+
host : ip-172-31-10-226.us-east-2.compute.internal
|
| 281 |
+
rank : 0 (local_rank: 0)
|
| 282 |
+
exitcode : 1 (pid: 464939)
|
| 283 |
+
error_file: <N/A>
|
| 284 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 285 |
+
============================================================
|
logs_backup/star_sweep_bs64_lr1e-5.log
ADDED
|
@@ -0,0 +1,284 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs64-lr1e-5', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 64, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 1e-05, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
|
| 2 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 3 |
+
return func(*args, **kwargs)
|
| 4 |
+
[rank0]:[W530 03:31:49.899347241 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 5 |
+
|
| 6 |
+
Running FSDP on rank = 0, world size = 1
|
| 7 |
+
/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
|
| 8 |
+
_init_core_state(
|
| 9 |
+
FullyShardedDataParallel(
|
| 10 |
+
(_fsdp_wrapped_module): Coconut(
|
| 11 |
+
(base_causallm): Qwen3ForCausalLM(
|
| 12 |
+
(model): Qwen3Model(
|
| 13 |
+
(embed_tokens): Embedding(151672, 1024)
|
| 14 |
+
(layers): ModuleList(
|
| 15 |
+
(0-27): 28 x FullyShardedDataParallel(
|
| 16 |
+
(_fsdp_wrapped_module): Qwen3DecoderLayer(
|
| 17 |
+
(self_attn): Qwen3Attention(
|
| 18 |
+
(q_proj): Linear(in_features=1024, out_features=2048, bias=False)
|
| 19 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 20 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 21 |
+
(o_proj): Linear(in_features=2048, out_features=1024, bias=False)
|
| 22 |
+
(q_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 23 |
+
(k_norm): Qwen3RMSNorm((128,), eps=1e-06)
|
| 24 |
+
)
|
| 25 |
+
(mlp): Qwen3MLP(
|
| 26 |
+
(gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 27 |
+
(up_proj): Linear(in_features=1024, out_features=3072, bias=False)
|
| 28 |
+
(down_proj): Linear(in_features=3072, out_features=1024, bias=False)
|
| 29 |
+
(act_fn): SiLUActivation()
|
| 30 |
+
)
|
| 31 |
+
(input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 32 |
+
(post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 33 |
+
)
|
| 34 |
+
)
|
| 35 |
+
)
|
| 36 |
+
(norm): Qwen3RMSNorm((1024,), eps=1e-06)
|
| 37 |
+
(rotary_emb): Qwen3RotaryEmbedding()
|
| 38 |
+
)
|
| 39 |
+
(lm_head): Linear(in_features=1024, out_features=151936, bias=False)
|
| 40 |
+
)
|
| 41 |
+
(embedding): Embedding(151672, 1024)
|
| 42 |
+
)
|
| 43 |
+
)
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
|
| 47 |
+
wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 48 |
+
wandb: Tracking run with wandb version 0.25.1
|
| 49 |
+
wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033216-nwpqaes3
|
| 50 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 51 |
+
wandb: Syncing run star-sweep-bs64-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-c9ba8dcb
|
| 52 |
+
wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
|
| 53 |
+
wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/nwpqaes3
|
| 54 |
+
|
| 55 |
+
============================================================
|
| 56 |
+
EPOCH 0/30 (thought_stage=1, data_stage=1)
|
| 57 |
+
============================================================
|
| 58 |
+
thought_stage=1, c_thought=1, max_difficulty=1
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
LR schedule: cosine, warmup=7 steps, total=4680 steps (max_steps/epoch=156)
|
| 65 |
+
|
| 66 |
+
wandb: WARNING Serializing object of type str that is 838345 bytes
|
| 67 |
+
Traceback (most recent call last):
|
| 68 |
+
File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 69 |
+
main()
|
| 70 |
+
File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 71 |
+
outputs = parallel_model(**batch)
|
| 72 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 73 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 74 |
+
return self._call_impl(*args, **kwargs)
|
| 75 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 76 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 77 |
+
return forward_call(*args, **kwargs)
|
| 78 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 79 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 80 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 81 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 82 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 83 |
+
return self._call_impl(*args, **kwargs)
|
| 84 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 85 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 86 |
+
return forward_call(*args, **kwargs)
|
| 87 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 88 |
+
File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
|
| 89 |
+
outputs = self.base_causallm(
|
| 90 |
+
^^^^^^^^^^^^^^^^^^^
|
| 91 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 92 |
+
return self._call_impl(*args, **kwargs)
|
| 93 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 94 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 95 |
+
return forward_call(*args, **kwargs)
|
| 96 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 97 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
|
| 98 |
+
output = func(self, *args, **kwargs)
|
| 99 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 100 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
|
| 101 |
+
outputs: BaseModelOutputWithPast = self.model(
|
| 102 |
+
^^^^^^^^^^^
|
| 103 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 104 |
+
return self._call_impl(*args, **kwargs)
|
| 105 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 106 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 107 |
+
return forward_call(*args, **kwargs)
|
| 108 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 109 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
|
| 110 |
+
output = func(self, *args, **kwargs)
|
| 111 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 112 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
|
| 113 |
+
outputs = func(self, *args, **kwargs)
|
| 114 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 115 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
|
| 116 |
+
hidden_states = decoder_layer(
|
| 117 |
+
^^^^^^^^^^^^^^
|
| 118 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 119 |
+
return self._call_impl(*args, **kwargs)
|
| 120 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 121 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 122 |
+
return forward_call(*args, **kwargs)
|
| 123 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 124 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 125 |
+
output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 126 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 127 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
|
| 128 |
+
return super().__call__(*args, **kwargs)
|
| 129 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 130 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 131 |
+
return self._call_impl(*args, **kwargs)
|
| 132 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 133 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 134 |
+
return inner()
|
| 135 |
+
^^^^^^^
|
| 136 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 137 |
+
result = forward_call(*args, **kwargs)
|
| 138 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 139 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 318, in forward
|
| 140 |
+
hidden_states, _ = self.self_attn(
|
| 141 |
+
^^^^^^^^^^^^^^^
|
| 142 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 143 |
+
return self._call_impl(*args, **kwargs)
|
| 144 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 145 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 146 |
+
return inner()
|
| 147 |
+
^^^^^^^
|
| 148 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 149 |
+
result = forward_call(*args, **kwargs)
|
| 150 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 151 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 277, in forward
|
| 152 |
+
attn_output, attn_weights = attention_interface(
|
| 153 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 154 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/integrations/sdpa_attention.py", line 92, in sdpa_attention_forward
|
| 155 |
+
attn_output = torch.nn.functional.scaled_dot_product_attention(
|
| 156 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 157 |
+
torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 406.00 MiB. GPU 0 has a total capacity of 139.80 GiB of which 45.06 MiB is free. Including non-PyTorch memory, this process has 139.75 GiB memory in use. Of the allocated memory 137.33 GiB is allocated by PyTorch, and 1.22 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 158 |
+
[rank0]: Traceback (most recent call last):
|
| 159 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
|
| 160 |
+
[rank0]: main()
|
| 161 |
+
[rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
|
| 162 |
+
[rank0]: outputs = parallel_model(**batch)
|
| 163 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
|
| 164 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 165 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 166 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 167 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 168 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 169 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 170 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 171 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 172 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 173 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 174 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 175 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 176 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 177 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 178 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 179 |
+
[rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
|
| 180 |
+
[rank0]: outputs = self.base_causallm(
|
| 181 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^
|
| 182 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 183 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 184 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 185 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 186 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 187 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 188 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
|
| 189 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 190 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 191 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
|
| 192 |
+
[rank0]: outputs: BaseModelOutputWithPast = self.model(
|
| 193 |
+
[rank0]: ^^^^^^^^^^^
|
| 194 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 195 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 196 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 197 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 198 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 199 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 200 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
|
| 201 |
+
[rank0]: output = func(self, *args, **kwargs)
|
| 202 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 203 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
|
| 204 |
+
[rank0]: outputs = func(self, *args, **kwargs)
|
| 205 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 206 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
|
| 207 |
+
[rank0]: hidden_states = decoder_layer(
|
| 208 |
+
[rank0]: ^^^^^^^^^^^^^^
|
| 209 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 210 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 211 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 212 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
|
| 213 |
+
[rank0]: return forward_call(*args, **kwargs)
|
| 214 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 215 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
|
| 216 |
+
[rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
|
| 217 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 218 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
|
| 219 |
+
[rank0]: return super().__call__(*args, **kwargs)
|
| 220 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 221 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 222 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 223 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 224 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 225 |
+
[rank0]: return inner()
|
| 226 |
+
[rank0]: ^^^^^^^
|
| 227 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 228 |
+
[rank0]: result = forward_call(*args, **kwargs)
|
| 229 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 230 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 318, in forward
|
| 231 |
+
[rank0]: hidden_states, _ = self.self_attn(
|
| 232 |
+
[rank0]: ^^^^^^^^^^^^^^^
|
| 233 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
|
| 234 |
+
[rank0]: return self._call_impl(*args, **kwargs)
|
| 235 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 236 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
|
| 237 |
+
[rank0]: return inner()
|
| 238 |
+
[rank0]: ^^^^^^^
|
| 239 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
|
| 240 |
+
[rank0]: result = forward_call(*args, **kwargs)
|
| 241 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 242 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 277, in forward
|
| 243 |
+
[rank0]: attn_output, attn_weights = attention_interface(
|
| 244 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^
|
| 245 |
+
[rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/integrations/sdpa_attention.py", line 92, in sdpa_attention_forward
|
| 246 |
+
[rank0]: attn_output = torch.nn.functional.scaled_dot_product_attention(
|
| 247 |
+
[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 248 |
+
[rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 406.00 MiB. GPU 0 has a total capacity of 139.80 GiB of which 45.06 MiB is free. Including non-PyTorch memory, this process has 139.75 GiB memory in use. Of the allocated memory 137.33 GiB is allocated by PyTorch, and 1.22 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
|
| 249 |
+
[1;34mwandb[0m:
|
| 250 |
+
[1;34mwandb[0m: 🚀 View run [33mstar-sweep-bs64-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-c9ba8dcb[0m at: [34mhttps://wandb.ai/seyedparsa/thoughtformer/runs/nwpqaes3[0m
|
| 251 |
+
[1;34mwandb[0m: Find logs at: [1;35mwandb/run-20260530_033216-nwpqaes3/logs[0m
|
| 252 |
+
E0530 03:32:27.283000 464320 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 464399) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
|
| 253 |
+
Traceback (most recent call last):
|
| 254 |
+
File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
|
| 255 |
+
sys.exit(main())
|
| 256 |
+
^^^^^^
|
| 257 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
|
| 258 |
+
return f(*args, **kwargs)
|
| 259 |
+
^^^^^^^^^^^^^^^^^^
|
| 260 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
|
| 261 |
+
run(args)
|
| 262 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
|
| 263 |
+
elastic_launch(
|
| 264 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
|
| 265 |
+
return launch_agent(self._config, self._entrypoint, list(args))
|
| 266 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 267 |
+
File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
|
| 268 |
+
raise ChildFailedError(
|
| 269 |
+
torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
|
| 270 |
+
============================================================
|
| 271 |
+
run.py FAILED
|
| 272 |
+
------------------------------------------------------------
|
| 273 |
+
Failures:
|
| 274 |
+
<NO_OTHER_FAILURES>
|
| 275 |
+
------------------------------------------------------------
|
| 276 |
+
Root Cause (first observed failure):
|
| 277 |
+
[0]:
|
| 278 |
+
time : 2026-05-30_03:32:27
|
| 279 |
+
host : ip-172-31-10-226.us-east-2.compute.internal
|
| 280 |
+
rank : 0 (local_rank: 0)
|
| 281 |
+
exitcode : 1 (pid: 464399)
|
| 282 |
+
error_file: <N/A>
|
| 283 |
+
traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
|
| 284 |
+
============================================================
|
logs_backup/star_thoughtformer.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|