seyedparsa commited on
Commit
ec1232d
·
verified ·
1 Parent(s): d84838e

backup: training logs before server expiry

Browse files
Files changed (44) hide show
  1. .gitattributes +7 -0
  2. logs_backup/diag_mem_bs4.log +0 -0
  3. logs_backup/star_atc_4gpu_fixed.log +0 -0
  4. logs_backup/star_atc_8gpu.log +3 -0
  5. logs_backup/star_atc_adaptive.log +0 -0
  6. logs_backup/star_atc_augmented.log +0 -0
  7. logs_backup/star_atc_best.log +0 -0
  8. logs_backup/star_atc_best_augmented.log +0 -0
  9. logs_backup/star_atc_k1L100.log +0 -0
  10. logs_backup/star_atc_k1L100_adaptive.log +3 -0
  11. logs_backup/star_atc_twoptr_repro.log +0 -0
  12. logs_backup/star_coconut_4gpu.log +3 -0
  13. logs_backup/star_coconut_k10L10_fixed.log +3 -0
  14. logs_backup/star_datacurr_4gpu_fixed.log +0 -0
  15. logs_backup/star_h2h_atc.log +0 -0
  16. logs_backup/star_h2h_nocot.log +0 -0
  17. logs_backup/star_k5_adaptive.log +0 -0
  18. logs_backup/star_k5_augmented.log +0 -0
  19. logs_backup/star_k5_no_thought.log +0 -0
  20. logs_backup/star_lr_1e-4.log +0 -0
  21. logs_backup/star_lr_1e-5.log +0 -0
  22. logs_backup/star_lr_2e-5.log +0 -0
  23. logs_backup/star_lr_5e-5.log +0 -0
  24. logs_backup/star_n1000_adaptive.log +0 -0
  25. logs_backup/star_n200_adaptive.log +0 -0
  26. logs_backup/star_n200_augmented.log +0 -0
  27. logs_backup/star_no_cot_repro.log +0 -0
  28. logs_backup/star_no_thought.log +0 -0
  29. logs_backup/star_nocot_4gpu_fixed.log +0 -0
  30. logs_backup/star_nocot_dc_k1L100.log +3 -0
  31. logs_backup/star_nocot_dc_lr1e5.log +0 -0
  32. logs_backup/star_nocot_nocurr_4gpu.log +3 -0
  33. logs_backup/star_nocot_nocurr_k1L100.log +3 -0
  34. logs_backup/star_nocot_nocurr_lr1e5.log +0 -0
  35. logs_backup/star_smoke.log +119 -0
  36. logs_backup/star_sweep_bs128_lr1e-4.log +296 -0
  37. logs_backup/star_sweep_bs128_lr1e-5.log +296 -0
  38. logs_backup/star_sweep_bs256_lr1e-4.log +290 -0
  39. logs_backup/star_sweep_bs256_lr1e-5.log +290 -0
  40. logs_backup/star_sweep_bs32_lr1e-4.log +152 -0
  41. logs_backup/star_sweep_bs32_lr1e-5.log +152 -0
  42. logs_backup/star_sweep_bs64_lr1e-4.log +285 -0
  43. logs_backup/star_sweep_bs64_lr1e-5.log +284 -0
  44. logs_backup/star_thoughtformer.log +0 -0
.gitattributes CHANGED
@@ -902,3 +902,10 @@ checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_n
902
  checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_no-reset_epoch-334c5641/train_state_30 filter=lfs diff=lfs merge=lfs -text
903
  checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_no-reset_epoch-334c5641/checkpoint_31 filter=lfs diff=lfs merge=lfs -text
904
  checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_no-reset_epoch-334c5641/train_state_31 filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
902
  checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_no-reset_epoch-334c5641/train_state_30 filter=lfs diff=lfs merge=lfs -text
903
  checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_no-reset_epoch-334c5641/checkpoint_31 filter=lfs diff=lfs merge=lfs -text
904
  checkpoints/star-coconut-k10L10-fixed-128_n1000_qwen3-0.6b-base_coconut_lr1e-4_no-reset_epoch-334c5641/train_state_31 filter=lfs diff=lfs merge=lfs -text
905
+ logs_backup/star_atc_8gpu.log filter=lfs diff=lfs merge=lfs -text
906
+ logs_backup/star_atc_k1L100_adaptive.log filter=lfs diff=lfs merge=lfs -text
907
+ logs_backup/star_coconut_4gpu.log filter=lfs diff=lfs merge=lfs -text
908
+ logs_backup/star_coconut_k10L10_fixed.log filter=lfs diff=lfs merge=lfs -text
909
+ logs_backup/star_nocot_dc_k1L100.log filter=lfs diff=lfs merge=lfs -text
910
+ logs_backup/star_nocot_nocurr_4gpu.log filter=lfs diff=lfs merge=lfs -text
911
+ logs_backup/star_nocot_nocurr_k1L100.log filter=lfs diff=lfs merge=lfs -text
logs_backup/diag_mem_bs4.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_atc_4gpu_fixed.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_atc_8gpu.log ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a895d341afb2c0c7e26ff3ff913350a1fb05fca0de77269330548c9de7512272
3
+ size 24023972
logs_backup/star_atc_adaptive.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_atc_augmented.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_atc_best.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_atc_best_augmented.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_atc_k1L100.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_atc_k1L100_adaptive.log ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:806b91d45db3b67ae77b3133beada72d506f3cb90816776af0af70b8d12d23a2
3
+ size 14951470
logs_backup/star_atc_twoptr_repro.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_coconut_4gpu.log ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5e5e5cc30c7f5a697b9d5d19172900cf4b90c8b0a028a6c90983761dc9f474b4
3
+ size 17755610
logs_backup/star_coconut_k10L10_fixed.log ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:08f7989e32e077f0c04fa2fea732fdf98c5b75501d4ea9cc8f3ba2c5614cc178
3
+ size 17668432
logs_backup/star_datacurr_4gpu_fixed.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_h2h_atc.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_h2h_nocot.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_k5_adaptive.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_k5_augmented.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_k5_no_thought.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_lr_1e-4.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_lr_1e-5.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_lr_2e-5.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_lr_5e-5.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_n1000_adaptive.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_n200_adaptive.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_n200_augmented.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_no_cot_repro.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_no_thought.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_nocot_4gpu_fixed.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_nocot_dc_k1L100.log ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:74e40e7fd0a0e606cc794c400df81a09a622de74aff7cfd02e6a4c9bab495446
3
+ size 13442075
logs_backup/star_nocot_dc_lr1e5.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_nocot_nocurr_4gpu.log ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2357a9bccacff1e8e7a7a86e87fa310af6bf47c3369a6ead5a2929c538639eb8
3
+ size 37777886
logs_backup/star_nocot_nocurr_k1L100.log ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f30faeed2612df03f87580ae5d59b74c1bf76230bbdfd4c384117107438d2121
3
+ size 24073355
logs_backup/star_nocot_nocurr_lr1e5.log ADDED
The diff for this file is too large to render. See raw diff
 
logs_backup/star_smoke.log ADDED
@@ -0,0 +1,119 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Config: {'project': 'thoughtformer', 'name': 'star-smoke', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': False, 'staging': 'two_pointers', 'init_thought_stage': 0, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 0, 'max_latent_stage': 0, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 4, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 2, 'lr': 0.0001, 'weight_decay': 0.01, 'group': 'smoke', 'tags': ['star', 'k10', 'L10', 'no_cot'], 'run_type': 'pilot'}
2
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ return func(*args, **kwargs)
4
+ [rank0]:[W528 07:02:16.327385348 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
5
+
6
+ Running FSDP on rank = 0, world size = 1
7
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
8
+ _init_core_state(
9
+ FullyShardedDataParallel(
10
+ (_fsdp_wrapped_module): Coconut(
11
+ (base_causallm): Qwen3ForCausalLM(
12
+ (model): Qwen3Model(
13
+ (embed_tokens): Embedding(151672, 1024)
14
+ (layers): ModuleList(
15
+ (0-27): 28 x FullyShardedDataParallel(
16
+ (_fsdp_wrapped_module): Qwen3DecoderLayer(
17
+ (self_attn): Qwen3Attention(
18
+ (q_proj): Linear(in_features=1024, out_features=2048, bias=False)
19
+ (k_proj): Linear(in_features=1024, out_features=1024, bias=False)
20
+ (v_proj): Linear(in_features=1024, out_features=1024, bias=False)
21
+ (o_proj): Linear(in_features=2048, out_features=1024, bias=False)
22
+ (q_norm): Qwen3RMSNorm((128,), eps=1e-06)
23
+ (k_norm): Qwen3RMSNorm((128,), eps=1e-06)
24
+ )
25
+ (mlp): Qwen3MLP(
26
+ (gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
27
+ (up_proj): Linear(in_features=1024, out_features=3072, bias=False)
28
+ (down_proj): Linear(in_features=3072, out_features=1024, bias=False)
29
+ (act_fn): SiLUActivation()
30
+ )
31
+ (input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
32
+ (post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
33
+ )
34
+ )
35
+ )
36
+ (norm): Qwen3RMSNorm((1024,), eps=1e-06)
37
+ (rotary_emb): Qwen3RotaryEmbedding()
38
+ )
39
+ (lm_head): Linear(in_features=1024, out_features=151936, bias=False)
40
+ )
41
+ (embedding): Embedding(151672, 1024)
42
+ )
43
+ )
44
+
45
+
46
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
47
+ wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
48
+ wandb: setting up run kbjpqcuk
49
+ wandb: Tracking run with wandb version 0.25.1
50
+ wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260528_070253-kbjpqcuk
51
+ wandb: Run `wandb offline` to turn off syncing.
52
+ wandb: Syncing run star-smoke_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_no-tc_two_pointers-bdad6776
53
+ wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
54
+ wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/kbjpqcuk
55
+
56
+ ============================================================
57
+ EPOCH 0/2 (thought_stage=0, data_stage=1)
58
+ ============================================================
59
+ thought_stage=0, c_thought=0, max_difficulty=1
60
+
61
+
62
+
63
+
64
+
65
+ LR schedule: cosine, warmup=125 steps, total=5000 steps (max_steps/epoch=2500)
66
+
67
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/autograd/graph.py:882: UserWarning: The AccumulateGrad node's stream does not match the stream of the node that produced the incoming gradient. This may incur unnecessary synchronization and break CUDA graph capture if the AccumulateGrad node's stream is the default stream. This mismatch is caused by an AccumulateGrad node created prior to the current iteration being kept alive. This can happen if the autograd graph is still being kept alive by tensors such as the loss, or if you are using DDP, which will stash a reference to the node. To resolve the mismatch, delete all references to the autograd graph or ensure that DDP initialization is performed under the same stream as subsequent forwards. If the mismatch is intentional, you can use torch.autograd.graph.set_warn_on_accumulate_grad_stream_mismatch(False) to suppress this warning. (Triggered internally at /pytorch/torch/csrc/autograd/input_buffer.cpp:240.)
68
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
69
+ rank 0 | gpu 0 | alloc=6.35 GB | reserved=27.88 GB | max=21.84 GB
70
+
71
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
72
+ return func(*args, **kwargs)
73
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:821: FutureWarning: FSDP.state_dict_type() and FSDP.set_state_dict_type() are being deprecated. Please use APIs, get_state_dict() and set_state_dict(), which can support different parallelisms, FSDP1, FSDP2, DDP. API doc: https://pytorch.org/docs/stable/distributed.checkpoint.html#torch.distributed.checkpoint.state_dict.get_state_dict .Tutorial: https://pytorch.org/tutorials/recipes/distributed_checkpoint_recipe.html .
74
+ prev_state_dict_settings = FullyShardedDataParallel.set_state_dict_type(
75
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/utils/_contextlib.py:124: UserWarning: When using ``NO_SHARD`` for ``ShardingStrategy``, full_state_dict will be returned.
76
+ return func(*args, **kwargs)
77
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/_optim_utils.py:1172: UserWarning: `_get_pg_default_device` will be deprecated, it only stays for backward-compatibility reason. If you need to find a device for object collectives, please use `_get_object_coll_device`. If you need to query the device types supported by group, please use `_device_capability(group)`.
78
+ device = _get_pg_default_device(group)
79
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:828: FutureWarning: FSDP.state_dict_type() and FSDP.set_state_dict_type() are being deprecated. Please use APIs, get_state_dict() and set_state_dict(), which can support different parallelisms, FSDP1, FSDP2, DDP. API doc: https://pytorch.org/docs/stable/distributed.checkpoint.html#torch.distributed.checkpoint.state_dict.get_state_dict .Tutorial: https://pytorch.org/tutorials/recipes/distributed_checkpoint_recipe.html .
80
+ FullyShardedDataParallel.set_state_dict_type(
81
+ saving model + train state.
82
+ Warning: failed to upload checkpoint_1 to HF: (Request ID: Root=1-6a17e919-1dce5dca1d2ddb050a1a0a4f;c5ffd1eb-6364-4939-8651-3bc2bb1d3fe2)
83
+
84
+ 403 Forbidden: You need to setup automatic credit recharge in order to upload more data. You can do so at /settings/billing..
85
+ Cannot access content at: https://huggingface.co/seyedparsa/thoughtformer-checkpoints.git/info/lfs/objects/batch.
86
+ Make sure your token has the correct permissions.
87
+ Warning: failed to upload train_state_1 to HF: (Request ID: Root=1-6a17e91d-3685080228a0e1cb2da99c67;496f22d9-c5af-4e21-bb1b-4dc6beb52df0)
88
+
89
+ 403 Forbidden: You need to setup automatic credit recharge in order to upload more data. You can do so at /settings/billing..
90
+ Cannot access content at: https://huggingface.co/seyedparsa/thoughtformer-checkpoints.git/info/lfs/objects/batch.
91
+ Make sure your token has the correct permissions.
92
+ Keeping all local checkpoints because HF upload failed; disk fallback until HF recovers.
93
+ eval loss 0.896042138338089
94
+
95
+ [WRONG] Example 0 (diff=3)
96
+ Expected : 'Patrick'
97
+ Extracted: 'Nancy'
98
+ Generated: '### Nancy'
99
+
100
+ [CORRECT] Example 1 (diff=2)
101
+ Expected : 'Roman'
102
+ Extracted: 'Roman'
103
+ Generated: '### Roman'
104
+
105
+ [WRONG] Example 2 (diff=2)
106
+ Expected : 'Harold'
107
+ Extracted: 'Stella'
108
+ Generated: '### Stella'
109
+
110
+ [WRONG] Example 3 (diff=2)
111
+ Expected : 'Larry'
112
+ Extracted: 'Liam'
113
+ Generated: '### Liam'
114
+
115
+ [CORRECT] Example 4 (diff=2)
116
+ Expected : 'Israel'
117
+ Extracted: 'Israel'
118
+ Generated: '### Israel'
119
+
logs_backup/star_sweep_bs128_lr1e-4.log ADDED
@@ -0,0 +1,296 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs128-lr1e-4', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 128, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 0.0001, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
2
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ return func(*args, **kwargs)
4
+ [rank0]:[W530 03:31:55.296639842 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
5
+
6
+ Running FSDP on rank = 0, world size = 1
7
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
8
+ _init_core_state(
9
+ FullyShardedDataParallel(
10
+ (_fsdp_wrapped_module): Coconut(
11
+ (base_causallm): Qwen3ForCausalLM(
12
+ (model): Qwen3Model(
13
+ (embed_tokens): Embedding(151672, 1024)
14
+ (layers): ModuleList(
15
+ (0-27): 28 x FullyShardedDataParallel(
16
+ (_fsdp_wrapped_module): Qwen3DecoderLayer(
17
+ (self_attn): Qwen3Attention(
18
+ (q_proj): Linear(in_features=1024, out_features=2048, bias=False)
19
+ (k_proj): Linear(in_features=1024, out_features=1024, bias=False)
20
+ (v_proj): Linear(in_features=1024, out_features=1024, bias=False)
21
+ (o_proj): Linear(in_features=2048, out_features=1024, bias=False)
22
+ (q_norm): Qwen3RMSNorm((128,), eps=1e-06)
23
+ (k_norm): Qwen3RMSNorm((128,), eps=1e-06)
24
+ )
25
+ (mlp): Qwen3MLP(
26
+ (gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
27
+ (up_proj): Linear(in_features=1024, out_features=3072, bias=False)
28
+ (down_proj): Linear(in_features=3072, out_features=1024, bias=False)
29
+ (act_fn): SiLUActivation()
30
+ )
31
+ (input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
32
+ (post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
33
+ )
34
+ )
35
+ )
36
+ (norm): Qwen3RMSNorm((1024,), eps=1e-06)
37
+ (rotary_emb): Qwen3RotaryEmbedding()
38
+ )
39
+ (lm_head): Linear(in_features=1024, out_features=151936, bias=False)
40
+ )
41
+ (embedding): Embedding(151672, 1024)
42
+ )
43
+ )
44
+
45
+
46
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
47
+ wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
48
+ wandb: Tracking run with wandb version 0.25.1
49
+ wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033222-2vtfied7
50
+ wandb: Run `wandb offline` to turn off syncing.
51
+ wandb: Syncing run star-sweep-bs128-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-6c6fcb83
52
+ wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
53
+ wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/2vtfied7
54
+
55
+ ============================================================
56
+ EPOCH 0/30 (thought_stage=1, data_stage=1)
57
+ ============================================================
58
+ thought_stage=1, c_thought=1, max_difficulty=1
59
+
60
+
61
+
62
+
63
+
64
+ LR schedule: cosine, warmup=3 steps, total=2340 steps (max_steps/epoch=78)
65
+
66
+ wandb: WARNING Serializing object of type str that is 1676601 bytes
67
+ Traceback (most recent call last):
68
+ File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
69
+ main()
70
+ File "/home/ubuntu/thoughtformer/run.py", line 937, in main
71
+ outputs = parallel_model(**batch)
72
+ ^^^^^^^^^^^^^^^^^^^^^^^
73
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
74
+ return self._call_impl(*args, **kwargs)
75
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
76
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
77
+ return forward_call(*args, **kwargs)
78
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
79
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
80
+ output = self._fsdp_wrapped_module(*args, **kwargs)
81
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
82
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
83
+ return self._call_impl(*args, **kwargs)
84
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
85
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
86
+ return forward_call(*args, **kwargs)
87
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
88
+ File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
89
+ outputs = self.base_causallm(
90
+ ^^^^^^^^^^^^^^^^^^^
91
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
92
+ return self._call_impl(*args, **kwargs)
93
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
94
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
95
+ return forward_call(*args, **kwargs)
96
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
97
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
98
+ output = func(self, *args, **kwargs)
99
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
100
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
101
+ outputs: BaseModelOutputWithPast = self.model(
102
+ ^^^^^^^^^^^
103
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
104
+ return self._call_impl(*args, **kwargs)
105
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
106
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
107
+ return forward_call(*args, **kwargs)
108
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
109
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
110
+ output = func(self, *args, **kwargs)
111
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
112
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
113
+ outputs = func(self, *args, **kwargs)
114
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
115
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
116
+ hidden_states = decoder_layer(
117
+ ^^^^^^^^^^^^^^
118
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
119
+ return self._call_impl(*args, **kwargs)
120
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
121
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
122
+ return forward_call(*args, **kwargs)
123
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
124
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
125
+ output = self._fsdp_wrapped_module(*args, **kwargs)
126
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
127
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
128
+ return super().__call__(*args, **kwargs)
129
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
130
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
131
+ return self._call_impl(*args, **kwargs)
132
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
133
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
134
+ return inner()
135
+ ^^^^^^^
136
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
137
+ result = forward_call(*args, **kwargs)
138
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
139
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
140
+ hidden_states = self.mlp(hidden_states)
141
+ ^^^^^^^^^^^^^^^^^^^^^^^
142
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
143
+ return self._call_impl(*args, **kwargs)
144
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
145
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
146
+ return forward_call(*args, **kwargs)
147
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
148
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
149
+ down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
150
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
151
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
152
+ return self._call_impl(*args, **kwargs)
153
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
154
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
155
+ return forward_call(*args, **kwargs)
156
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
157
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/activations.py", line 103, in forward
158
+ return nn.functional.silu(input)
159
+ ^^^^^^^^^^^^^^^^^^^^^^^^^
160
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/functional.py", line 2397, in silu
161
+ return torch._C._nn.silu(input)
162
+ ^^^^^^^^^^^^^^^^^^^^^^^^
163
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 1.19 GiB. GPU 0 has a total capacity of 139.80 GiB of which 583.06 MiB is free. Including non-PyTorch memory, this process has 139.22 GiB memory in use. Of the allocated memory 136.99 GiB is allocated by PyTorch, and 1.03 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
164
+ [rank0]: Traceback (most recent call last):
165
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
166
+ [rank0]: main()
167
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
168
+ [rank0]: outputs = parallel_model(**batch)
169
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
170
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
171
+ [rank0]: return self._call_impl(*args, **kwargs)
172
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
173
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
174
+ [rank0]: return forward_call(*args, **kwargs)
175
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
176
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
177
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
178
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
179
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
180
+ [rank0]: return self._call_impl(*args, **kwargs)
181
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
182
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
183
+ [rank0]: return forward_call(*args, **kwargs)
184
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
185
+ [rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
186
+ [rank0]: outputs = self.base_causallm(
187
+ [rank0]: ^^^^^^^^^^^^^^^^^^^
188
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
189
+ [rank0]: return self._call_impl(*args, **kwargs)
190
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
191
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
192
+ [rank0]: return forward_call(*args, **kwargs)
193
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
194
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
195
+ [rank0]: output = func(self, *args, **kwargs)
196
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
197
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
198
+ [rank0]: outputs: BaseModelOutputWithPast = self.model(
199
+ [rank0]: ^^^^^^^^^^^
200
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
201
+ [rank0]: return self._call_impl(*args, **kwargs)
202
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
203
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
204
+ [rank0]: return forward_call(*args, **kwargs)
205
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
206
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
207
+ [rank0]: output = func(self, *args, **kwargs)
208
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
209
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
210
+ [rank0]: outputs = func(self, *args, **kwargs)
211
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
212
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
213
+ [rank0]: hidden_states = decoder_layer(
214
+ [rank0]: ^^^^^^^^^^^^^^
215
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
216
+ [rank0]: return self._call_impl(*args, **kwargs)
217
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
218
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
219
+ [rank0]: return forward_call(*args, **kwargs)
220
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
221
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
222
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
223
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
224
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
225
+ [rank0]: return super().__call__(*args, **kwargs)
226
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
227
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
228
+ [rank0]: return self._call_impl(*args, **kwargs)
229
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
230
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
231
+ [rank0]: return inner()
232
+ [rank0]: ^^^^^^^
233
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
234
+ [rank0]: result = forward_call(*args, **kwargs)
235
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
236
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
237
+ [rank0]: hidden_states = self.mlp(hidden_states)
238
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
239
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
240
+ [rank0]: return self._call_impl(*args, **kwargs)
241
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
242
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
243
+ [rank0]: return forward_call(*args, **kwargs)
244
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
245
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
246
+ [rank0]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
247
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
248
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
249
+ [rank0]: return self._call_impl(*args, **kwargs)
250
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
251
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
252
+ [rank0]: return forward_call(*args, **kwargs)
253
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
254
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/activations.py", line 103, in forward
255
+ [rank0]: return nn.functional.silu(input)
256
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^
257
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/functional.py", line 2397, in silu
258
+ [rank0]: return torch._C._nn.silu(input)
259
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^
260
+ [rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 1.19 GiB. GPU 0 has a total capacity of 139.80 GiB of which 583.06 MiB is free. Including non-PyTorch memory, this process has 139.22 GiB memory in use. Of the allocated memory 136.99 GiB is allocated by PyTorch, and 1.03 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
261
+ wandb:
262
+ wandb: 🚀 View run star-sweep-bs128-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-6c6fcb83 at: https://wandb.ai/seyedparsa/thoughtformer/runs/2vtfied7
263
+ wandb: Find logs at: wandb/run-20260530_033222-2vtfied7/logs
264
+ E0530 03:32:33.929000 465640 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 466218) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
265
+ Traceback (most recent call last):
266
+ File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
267
+ sys.exit(main())
268
+ ^^^^^^
269
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
270
+ return f(*args, **kwargs)
271
+ ^^^^^^^^^^^^^^^^^^
272
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
273
+ run(args)
274
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
275
+ elastic_launch(
276
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
277
+ return launch_agent(self._config, self._entrypoint, list(args))
278
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
279
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
280
+ raise ChildFailedError(
281
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
282
+ ============================================================
283
+ run.py FAILED
284
+ ------------------------------------------------------------
285
+ Failures:
286
+ <NO_OTHER_FAILURES>
287
+ ------------------------------------------------------------
288
+ Root Cause (first observed failure):
289
+ [0]:
290
+ time : 2026-05-30_03:32:33
291
+ host : ip-172-31-10-226.us-east-2.compute.internal
292
+ rank : 0 (local_rank: 0)
293
+ exitcode : 1 (pid: 466218)
294
+ error_file: <N/A>
295
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
296
+ ============================================================
logs_backup/star_sweep_bs128_lr1e-5.log ADDED
@@ -0,0 +1,296 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs128-lr1e-5', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 128, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 1e-05, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
2
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ return func(*args, **kwargs)
4
+ [rank0]:[W530 03:31:53.943261833 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
5
+
6
+ Running FSDP on rank = 0, world size = 1
7
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
8
+ _init_core_state(
9
+ FullyShardedDataParallel(
10
+ (_fsdp_wrapped_module): Coconut(
11
+ (base_causallm): Qwen3ForCausalLM(
12
+ (model): Qwen3Model(
13
+ (embed_tokens): Embedding(151672, 1024)
14
+ (layers): ModuleList(
15
+ (0-27): 28 x FullyShardedDataParallel(
16
+ (_fsdp_wrapped_module): Qwen3DecoderLayer(
17
+ (self_attn): Qwen3Attention(
18
+ (q_proj): Linear(in_features=1024, out_features=2048, bias=False)
19
+ (k_proj): Linear(in_features=1024, out_features=1024, bias=False)
20
+ (v_proj): Linear(in_features=1024, out_features=1024, bias=False)
21
+ (o_proj): Linear(in_features=2048, out_features=1024, bias=False)
22
+ (q_norm): Qwen3RMSNorm((128,), eps=1e-06)
23
+ (k_norm): Qwen3RMSNorm((128,), eps=1e-06)
24
+ )
25
+ (mlp): Qwen3MLP(
26
+ (gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
27
+ (up_proj): Linear(in_features=1024, out_features=3072, bias=False)
28
+ (down_proj): Linear(in_features=3072, out_features=1024, bias=False)
29
+ (act_fn): SiLUActivation()
30
+ )
31
+ (input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
32
+ (post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
33
+ )
34
+ )
35
+ )
36
+ (norm): Qwen3RMSNorm((1024,), eps=1e-06)
37
+ (rotary_emb): Qwen3RotaryEmbedding()
38
+ )
39
+ (lm_head): Linear(in_features=1024, out_features=151936, bias=False)
40
+ )
41
+ (embedding): Embedding(151672, 1024)
42
+ )
43
+ )
44
+
45
+
46
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
47
+ wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
48
+ wandb: Tracking run with wandb version 0.25.1
49
+ wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033220-efpd3uen
50
+ wandb: Run `wandb offline` to turn off syncing.
51
+ wandb: Syncing run star-sweep-bs128-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-46d2d3b7
52
+ wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
53
+ wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/efpd3uen
54
+
55
+ ============================================================
56
+ EPOCH 0/30 (thought_stage=1, data_stage=1)
57
+ ============================================================
58
+ thought_stage=1, c_thought=1, max_difficulty=1
59
+
60
+
61
+
62
+
63
+
64
+ LR schedule: cosine, warmup=3 steps, total=2340 steps (max_steps/epoch=78)
65
+
66
+ wandb: WARNING Serializing object of type str that is 1676601 bytes
67
+ Traceback (most recent call last):
68
+ File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
69
+ main()
70
+ File "/home/ubuntu/thoughtformer/run.py", line 937, in main
71
+ outputs = parallel_model(**batch)
72
+ ^^^^^^^^^^^^^^^^^^^^^^^
73
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
74
+ return self._call_impl(*args, **kwargs)
75
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
76
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
77
+ return forward_call(*args, **kwargs)
78
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
79
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
80
+ output = self._fsdp_wrapped_module(*args, **kwargs)
81
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
82
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
83
+ return self._call_impl(*args, **kwargs)
84
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
85
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
86
+ return forward_call(*args, **kwargs)
87
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
88
+ File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
89
+ outputs = self.base_causallm(
90
+ ^^^^^^^^^^^^^^^^^^^
91
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
92
+ return self._call_impl(*args, **kwargs)
93
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
94
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
95
+ return forward_call(*args, **kwargs)
96
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
97
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
98
+ output = func(self, *args, **kwargs)
99
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
100
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
101
+ outputs: BaseModelOutputWithPast = self.model(
102
+ ^^^^^^^^^^^
103
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
104
+ return self._call_impl(*args, **kwargs)
105
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
106
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
107
+ return forward_call(*args, **kwargs)
108
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
109
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
110
+ output = func(self, *args, **kwargs)
111
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
112
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
113
+ outputs = func(self, *args, **kwargs)
114
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
115
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
116
+ hidden_states = decoder_layer(
117
+ ^^^^^^^^^^^^^^
118
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
119
+ return self._call_impl(*args, **kwargs)
120
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
121
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
122
+ return forward_call(*args, **kwargs)
123
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
124
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
125
+ output = self._fsdp_wrapped_module(*args, **kwargs)
126
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
127
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
128
+ return super().__call__(*args, **kwargs)
129
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
130
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
131
+ return self._call_impl(*args, **kwargs)
132
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
133
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
134
+ return inner()
135
+ ^^^^^^^
136
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
137
+ result = forward_call(*args, **kwargs)
138
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
139
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
140
+ hidden_states = self.mlp(hidden_states)
141
+ ^^^^^^^^^^^^^^^^^^^^^^^
142
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
143
+ return self._call_impl(*args, **kwargs)
144
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
145
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
146
+ return forward_call(*args, **kwargs)
147
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
148
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
149
+ down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
150
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
151
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
152
+ return self._call_impl(*args, **kwargs)
153
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
154
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
155
+ return forward_call(*args, **kwargs)
156
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
157
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/activations.py", line 103, in forward
158
+ return nn.functional.silu(input)
159
+ ^^^^^^^^^^^^^^^^^^^^^^^^^
160
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/functional.py", line 2397, in silu
161
+ return torch._C._nn.silu(input)
162
+ ^^^^^^^^^^^^^^^^^^^^^^^^
163
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 1.19 GiB. GPU 0 has a total capacity of 139.80 GiB of which 583.06 MiB is free. Including non-PyTorch memory, this process has 139.22 GiB memory in use. Of the allocated memory 136.99 GiB is allocated by PyTorch, and 1.03 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
164
+ [rank0]: Traceback (most recent call last):
165
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
166
+ [rank0]: main()
167
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
168
+ [rank0]: outputs = parallel_model(**batch)
169
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
170
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
171
+ [rank0]: return self._call_impl(*args, **kwargs)
172
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
173
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
174
+ [rank0]: return forward_call(*args, **kwargs)
175
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
176
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
177
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
178
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
179
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
180
+ [rank0]: return self._call_impl(*args, **kwargs)
181
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
182
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
183
+ [rank0]: return forward_call(*args, **kwargs)
184
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
185
+ [rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
186
+ [rank0]: outputs = self.base_causallm(
187
+ [rank0]: ^^^^^^^^^^^^^^^^^^^
188
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
189
+ [rank0]: return self._call_impl(*args, **kwargs)
190
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
191
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
192
+ [rank0]: return forward_call(*args, **kwargs)
193
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
194
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
195
+ [rank0]: output = func(self, *args, **kwargs)
196
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
197
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
198
+ [rank0]: outputs: BaseModelOutputWithPast = self.model(
199
+ [rank0]: ^^^^^^^^^^^
200
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
201
+ [rank0]: return self._call_impl(*args, **kwargs)
202
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
203
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
204
+ [rank0]: return forward_call(*args, **kwargs)
205
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
206
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
207
+ [rank0]: output = func(self, *args, **kwargs)
208
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
209
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
210
+ [rank0]: outputs = func(self, *args, **kwargs)
211
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
212
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
213
+ [rank0]: hidden_states = decoder_layer(
214
+ [rank0]: ^^^^^^^^^^^^^^
215
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
216
+ [rank0]: return self._call_impl(*args, **kwargs)
217
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
218
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
219
+ [rank0]: return forward_call(*args, **kwargs)
220
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
221
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
222
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
223
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
224
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
225
+ [rank0]: return super().__call__(*args, **kwargs)
226
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
227
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
228
+ [rank0]: return self._call_impl(*args, **kwargs)
229
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
230
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
231
+ [rank0]: return inner()
232
+ [rank0]: ^^^^^^^
233
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
234
+ [rank0]: result = forward_call(*args, **kwargs)
235
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
236
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
237
+ [rank0]: hidden_states = self.mlp(hidden_states)
238
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
239
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
240
+ [rank0]: return self._call_impl(*args, **kwargs)
241
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
242
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
243
+ [rank0]: return forward_call(*args, **kwargs)
244
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
245
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
246
+ [rank0]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
247
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
248
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
249
+ [rank0]: return self._call_impl(*args, **kwargs)
250
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
251
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
252
+ [rank0]: return forward_call(*args, **kwargs)
253
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
254
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/activations.py", line 103, in forward
255
+ [rank0]: return nn.functional.silu(input)
256
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^
257
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/functional.py", line 2397, in silu
258
+ [rank0]: return torch._C._nn.silu(input)
259
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^
260
+ [rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 1.19 GiB. GPU 0 has a total capacity of 139.80 GiB of which 583.06 MiB is free. Including non-PyTorch memory, this process has 139.22 GiB memory in use. Of the allocated memory 136.99 GiB is allocated by PyTorch, and 1.03 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
261
+ wandb:
262
+ wandb: 🚀 View run star-sweep-bs128-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-46d2d3b7 at: https://wandb.ai/seyedparsa/thoughtformer/runs/efpd3uen
263
+ wandb: Find logs at: wandb/run-20260530_033220-efpd3uen/logs
264
+ E0530 03:32:31.760000 464943 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 465570) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
265
+ Traceback (most recent call last):
266
+ File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
267
+ sys.exit(main())
268
+ ^^^^^^
269
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
270
+ return f(*args, **kwargs)
271
+ ^^^^^^^^^^^^^^^^^^
272
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
273
+ run(args)
274
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
275
+ elastic_launch(
276
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
277
+ return launch_agent(self._config, self._entrypoint, list(args))
278
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
279
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
280
+ raise ChildFailedError(
281
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
282
+ ============================================================
283
+ run.py FAILED
284
+ ------------------------------------------------------------
285
+ Failures:
286
+ <NO_OTHER_FAILURES>
287
+ ------------------------------------------------------------
288
+ Root Cause (first observed failure):
289
+ [0]:
290
+ time : 2026-05-30_03:32:31
291
+ host : ip-172-31-10-226.us-east-2.compute.internal
292
+ rank : 0 (local_rank: 0)
293
+ exitcode : 1 (pid: 465570)
294
+ error_file: <N/A>
295
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
296
+ ============================================================
logs_backup/star_sweep_bs256_lr1e-4.log ADDED
@@ -0,0 +1,290 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs256-lr1e-4', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 256, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 0.0001, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
2
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ return func(*args, **kwargs)
4
+ [rank0]:[W530 03:31:59.134090576 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
5
+
6
+ Running FSDP on rank = 0, world size = 1
7
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
8
+ _init_core_state(
9
+ FullyShardedDataParallel(
10
+ (_fsdp_wrapped_module): Coconut(
11
+ (base_causallm): Qwen3ForCausalLM(
12
+ (model): Qwen3Model(
13
+ (embed_tokens): Embedding(151672, 1024)
14
+ (layers): ModuleList(
15
+ (0-27): 28 x FullyShardedDataParallel(
16
+ (_fsdp_wrapped_module): Qwen3DecoderLayer(
17
+ (self_attn): Qwen3Attention(
18
+ (q_proj): Linear(in_features=1024, out_features=2048, bias=False)
19
+ (k_proj): Linear(in_features=1024, out_features=1024, bias=False)
20
+ (v_proj): Linear(in_features=1024, out_features=1024, bias=False)
21
+ (o_proj): Linear(in_features=2048, out_features=1024, bias=False)
22
+ (q_norm): Qwen3RMSNorm((128,), eps=1e-06)
23
+ (k_norm): Qwen3RMSNorm((128,), eps=1e-06)
24
+ )
25
+ (mlp): Qwen3MLP(
26
+ (gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
27
+ (up_proj): Linear(in_features=1024, out_features=3072, bias=False)
28
+ (down_proj): Linear(in_features=3072, out_features=1024, bias=False)
29
+ (act_fn): SiLUActivation()
30
+ )
31
+ (input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
32
+ (post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
33
+ )
34
+ )
35
+ )
36
+ (norm): Qwen3RMSNorm((1024,), eps=1e-06)
37
+ (rotary_emb): Qwen3RotaryEmbedding()
38
+ )
39
+ (lm_head): Linear(in_features=1024, out_features=151936, bias=False)
40
+ )
41
+ (embedding): Embedding(151672, 1024)
42
+ )
43
+ )
44
+
45
+
46
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
47
+ wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
48
+ wandb: Tracking run with wandb version 0.25.1
49
+ wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033226-wj48phx5
50
+ wandb: Run `wandb offline` to turn off syncing.
51
+ wandb: Syncing run star-sweep-bs256-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-ff592230
52
+ wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
53
+ wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/wj48phx5
54
+
55
+ ============================================================
56
+ EPOCH 0/30 (thought_stage=1, data_stage=1)
57
+ ============================================================
58
+ thought_stage=1, c_thought=1, max_difficulty=1
59
+
60
+
61
+
62
+
63
+
64
+ LR schedule: cosine, warmup=1 steps, total=1170 steps (max_steps/epoch=39)
65
+
66
+ wandb: WARNING Serializing object of type str that is 3353666 bytes
67
+ Traceback (most recent call last):
68
+ File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
69
+ main()
70
+ File "/home/ubuntu/thoughtformer/run.py", line 937, in main
71
+ outputs = parallel_model(**batch)
72
+ ^^^^^^^^^^^^^^^^^^^^^^^
73
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
74
+ return self._call_impl(*args, **kwargs)
75
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
76
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
77
+ return forward_call(*args, **kwargs)
78
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
79
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
80
+ output = self._fsdp_wrapped_module(*args, **kwargs)
81
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
82
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
83
+ return self._call_impl(*args, **kwargs)
84
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
85
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
86
+ return forward_call(*args, **kwargs)
87
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
88
+ File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
89
+ outputs = self.base_causallm(
90
+ ^^^^^^^^^^^^^^^^^^^
91
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
92
+ return self._call_impl(*args, **kwargs)
93
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
94
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
95
+ return forward_call(*args, **kwargs)
96
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
97
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
98
+ output = func(self, *args, **kwargs)
99
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
100
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
101
+ outputs: BaseModelOutputWithPast = self.model(
102
+ ^^^^^^^^^^^
103
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
104
+ return self._call_impl(*args, **kwargs)
105
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
106
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
107
+ return forward_call(*args, **kwargs)
108
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
109
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
110
+ output = func(self, *args, **kwargs)
111
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
112
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
113
+ outputs = func(self, *args, **kwargs)
114
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
115
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
116
+ hidden_states = decoder_layer(
117
+ ^^^^^^^^^^^^^^
118
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
119
+ return self._call_impl(*args, **kwargs)
120
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
121
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
122
+ return forward_call(*args, **kwargs)
123
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
124
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
125
+ output = self._fsdp_wrapped_module(*args, **kwargs)
126
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
127
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
128
+ return super().__call__(*args, **kwargs)
129
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
130
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
131
+ return self._call_impl(*args, **kwargs)
132
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
133
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
134
+ return inner()
135
+ ^^^^^^^
136
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
137
+ result = forward_call(*args, **kwargs)
138
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
139
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
140
+ hidden_states = self.mlp(hidden_states)
141
+ ^^^^^^^^^^^^^^^^^^^^^^^
142
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
143
+ return self._call_impl(*args, **kwargs)
144
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
145
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
146
+ return forward_call(*args, **kwargs)
147
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
148
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
149
+ down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
150
+ ^^^^^^^^^^^^^^^
151
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
152
+ return self._call_impl(*args, **kwargs)
153
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
154
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
155
+ return forward_call(*args, **kwargs)
156
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
157
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/linear.py", line 134, in forward
158
+ return F.linear(input, self.weight, self.bias)
159
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
160
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 2.38 GiB. GPU 0 has a total capacity of 139.80 GiB of which 1.40 GiB is free. Including non-PyTorch memory, this process has 138.39 GiB memory in use. Of the allocated memory 135.99 GiB is allocated by PyTorch, and 1.20 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
161
+ [rank0]: Traceback (most recent call last):
162
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
163
+ [rank0]: main()
164
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
165
+ [rank0]: outputs = parallel_model(**batch)
166
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
167
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
168
+ [rank0]: return self._call_impl(*args, **kwargs)
169
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
170
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
171
+ [rank0]: return forward_call(*args, **kwargs)
172
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
173
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
174
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
175
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
176
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
177
+ [rank0]: return self._call_impl(*args, **kwargs)
178
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
179
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
180
+ [rank0]: return forward_call(*args, **kwargs)
181
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
182
+ [rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
183
+ [rank0]: outputs = self.base_causallm(
184
+ [rank0]: ^^^^^^^^^^^^^^^^^^^
185
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
186
+ [rank0]: return self._call_impl(*args, **kwargs)
187
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
188
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
189
+ [rank0]: return forward_call(*args, **kwargs)
190
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
191
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
192
+ [rank0]: output = func(self, *args, **kwargs)
193
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
194
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
195
+ [rank0]: outputs: BaseModelOutputWithPast = self.model(
196
+ [rank0]: ^^^^^^^^^^^
197
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
198
+ [rank0]: return self._call_impl(*args, **kwargs)
199
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
200
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
201
+ [rank0]: return forward_call(*args, **kwargs)
202
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
203
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
204
+ [rank0]: output = func(self, *args, **kwargs)
205
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
206
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
207
+ [rank0]: outputs = func(self, *args, **kwargs)
208
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
209
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
210
+ [rank0]: hidden_states = decoder_layer(
211
+ [rank0]: ^^^^^^^^^^^^^^
212
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
213
+ [rank0]: return self._call_impl(*args, **kwargs)
214
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
215
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
216
+ [rank0]: return forward_call(*args, **kwargs)
217
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
218
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
219
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
220
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
221
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
222
+ [rank0]: return super().__call__(*args, **kwargs)
223
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
224
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
225
+ [rank0]: return self._call_impl(*args, **kwargs)
226
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
227
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
228
+ [rank0]: return inner()
229
+ [rank0]: ^^^^^^^
230
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
231
+ [rank0]: result = forward_call(*args, **kwargs)
232
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
233
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
234
+ [rank0]: hidden_states = self.mlp(hidden_states)
235
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
236
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
237
+ [rank0]: return self._call_impl(*args, **kwargs)
238
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
239
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
240
+ [rank0]: return forward_call(*args, **kwargs)
241
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
242
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
243
+ [rank0]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
244
+ [rank0]: ^^^^^^^^^^^^^^^
245
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
246
+ [rank0]: return self._call_impl(*args, **kwargs)
247
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
248
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
249
+ [rank0]: return forward_call(*args, **kwargs)
250
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
251
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/linear.py", line 134, in forward
252
+ [rank0]: return F.linear(input, self.weight, self.bias)
253
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
254
+ [rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 2.38 GiB. GPU 0 has a total capacity of 139.80 GiB of which 1.40 GiB is free. Including non-PyTorch memory, this process has 138.39 GiB memory in use. Of the allocated memory 135.99 GiB is allocated by PyTorch, and 1.20 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
255
+ wandb:
256
+ wandb: 🚀 View run star-sweep-bs256-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-ff592230 at: https://wandb.ai/seyedparsa/thoughtformer/runs/wj48phx5
257
+ wandb: Find logs at: wandb/run-20260530_033226-wj48phx5/logs
258
+ E0530 03:32:40.146000 466881 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 467529) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
259
+ Traceback (most recent call last):
260
+ File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
261
+ sys.exit(main())
262
+ ^^^^^^
263
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
264
+ return f(*args, **kwargs)
265
+ ^^^^^^^^^^^^^^^^^^
266
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
267
+ run(args)
268
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
269
+ elastic_launch(
270
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
271
+ return launch_agent(self._config, self._entrypoint, list(args))
272
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
273
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
274
+ raise ChildFailedError(
275
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
276
+ ============================================================
277
+ run.py FAILED
278
+ ------------------------------------------------------------
279
+ Failures:
280
+ <NO_OTHER_FAILURES>
281
+ ------------------------------------------------------------
282
+ Root Cause (first observed failure):
283
+ [0]:
284
+ time : 2026-05-30_03:32:40
285
+ host : ip-172-31-10-226.us-east-2.compute.internal
286
+ rank : 0 (local_rank: 0)
287
+ exitcode : 1 (pid: 467529)
288
+ error_file: <N/A>
289
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
290
+ ============================================================
logs_backup/star_sweep_bs256_lr1e-5.log ADDED
@@ -0,0 +1,290 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs256-lr1e-5', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 256, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 1e-05, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
2
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ return func(*args, **kwargs)
4
+ [rank0]:[W530 03:31:57.089584891 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
5
+
6
+ Running FSDP on rank = 0, world size = 1
7
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
8
+ _init_core_state(
9
+ FullyShardedDataParallel(
10
+ (_fsdp_wrapped_module): Coconut(
11
+ (base_causallm): Qwen3ForCausalLM(
12
+ (model): Qwen3Model(
13
+ (embed_tokens): Embedding(151672, 1024)
14
+ (layers): ModuleList(
15
+ (0-27): 28 x FullyShardedDataParallel(
16
+ (_fsdp_wrapped_module): Qwen3DecoderLayer(
17
+ (self_attn): Qwen3Attention(
18
+ (q_proj): Linear(in_features=1024, out_features=2048, bias=False)
19
+ (k_proj): Linear(in_features=1024, out_features=1024, bias=False)
20
+ (v_proj): Linear(in_features=1024, out_features=1024, bias=False)
21
+ (o_proj): Linear(in_features=2048, out_features=1024, bias=False)
22
+ (q_norm): Qwen3RMSNorm((128,), eps=1e-06)
23
+ (k_norm): Qwen3RMSNorm((128,), eps=1e-06)
24
+ )
25
+ (mlp): Qwen3MLP(
26
+ (gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
27
+ (up_proj): Linear(in_features=1024, out_features=3072, bias=False)
28
+ (down_proj): Linear(in_features=3072, out_features=1024, bias=False)
29
+ (act_fn): SiLUActivation()
30
+ )
31
+ (input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
32
+ (post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
33
+ )
34
+ )
35
+ )
36
+ (norm): Qwen3RMSNorm((1024,), eps=1e-06)
37
+ (rotary_emb): Qwen3RotaryEmbedding()
38
+ )
39
+ (lm_head): Linear(in_features=1024, out_features=151936, bias=False)
40
+ )
41
+ (embedding): Embedding(151672, 1024)
42
+ )
43
+ )
44
+
45
+
46
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
47
+ wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
48
+ wandb: Tracking run with wandb version 0.25.1
49
+ wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033224-tjrts1mq
50
+ wandb: Run `wandb offline` to turn off syncing.
51
+ wandb: Syncing run star-sweep-bs256-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-a9a8c070
52
+ wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
53
+ wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/tjrts1mq
54
+
55
+ ============================================================
56
+ EPOCH 0/30 (thought_stage=1, data_stage=1)
57
+ ============================================================
58
+ thought_stage=1, c_thought=1, max_difficulty=1
59
+
60
+
61
+
62
+
63
+
64
+ LR schedule: cosine, warmup=1 steps, total=1170 steps (max_steps/epoch=39)
65
+
66
+ wandb: WARNING Serializing object of type str that is 3353666 bytes
67
+ Traceback (most recent call last):
68
+ File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
69
+ main()
70
+ File "/home/ubuntu/thoughtformer/run.py", line 937, in main
71
+ outputs = parallel_model(**batch)
72
+ ^^^^^^^^^^^^^^^^^^^^^^^
73
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
74
+ return self._call_impl(*args, **kwargs)
75
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
76
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
77
+ return forward_call(*args, **kwargs)
78
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
79
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
80
+ output = self._fsdp_wrapped_module(*args, **kwargs)
81
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
82
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
83
+ return self._call_impl(*args, **kwargs)
84
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
85
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
86
+ return forward_call(*args, **kwargs)
87
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
88
+ File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
89
+ outputs = self.base_causallm(
90
+ ^^^^^^^^^^^^^^^^^^^
91
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
92
+ return self._call_impl(*args, **kwargs)
93
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
94
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
95
+ return forward_call(*args, **kwargs)
96
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
97
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
98
+ output = func(self, *args, **kwargs)
99
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
100
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
101
+ outputs: BaseModelOutputWithPast = self.model(
102
+ ^^^^^^^^^^^
103
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
104
+ return self._call_impl(*args, **kwargs)
105
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
106
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
107
+ return forward_call(*args, **kwargs)
108
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
109
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
110
+ output = func(self, *args, **kwargs)
111
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
112
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
113
+ outputs = func(self, *args, **kwargs)
114
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
115
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
116
+ hidden_states = decoder_layer(
117
+ ^^^^^^^^^^^^^^
118
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
119
+ return self._call_impl(*args, **kwargs)
120
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
121
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
122
+ return forward_call(*args, **kwargs)
123
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
124
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
125
+ output = self._fsdp_wrapped_module(*args, **kwargs)
126
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
127
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
128
+ return super().__call__(*args, **kwargs)
129
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
130
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
131
+ return self._call_impl(*args, **kwargs)
132
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
133
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
134
+ return inner()
135
+ ^^^^^^^
136
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
137
+ result = forward_call(*args, **kwargs)
138
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
139
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
140
+ hidden_states = self.mlp(hidden_states)
141
+ ^^^^^^^^^^^^^^^^^^^^^^^
142
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
143
+ return self._call_impl(*args, **kwargs)
144
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
145
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
146
+ return forward_call(*args, **kwargs)
147
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
148
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
149
+ down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
150
+ ^^^^^^^^^^^^^^^
151
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
152
+ return self._call_impl(*args, **kwargs)
153
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
154
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
155
+ return forward_call(*args, **kwargs)
156
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
157
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/linear.py", line 134, in forward
158
+ return F.linear(input, self.weight, self.bias)
159
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
160
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 2.38 GiB. GPU 0 has a total capacity of 139.80 GiB of which 1.40 GiB is free. Including non-PyTorch memory, this process has 138.39 GiB memory in use. Of the allocated memory 135.99 GiB is allocated by PyTorch, and 1.20 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
161
+ [rank0]: Traceback (most recent call last):
162
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
163
+ [rank0]: main()
164
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
165
+ [rank0]: outputs = parallel_model(**batch)
166
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
167
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
168
+ [rank0]: return self._call_impl(*args, **kwargs)
169
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
170
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
171
+ [rank0]: return forward_call(*args, **kwargs)
172
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
173
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
174
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
175
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
176
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
177
+ [rank0]: return self._call_impl(*args, **kwargs)
178
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
179
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
180
+ [rank0]: return forward_call(*args, **kwargs)
181
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
182
+ [rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
183
+ [rank0]: outputs = self.base_causallm(
184
+ [rank0]: ^^^^^^^^^^^^^^^^^^^
185
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
186
+ [rank0]: return self._call_impl(*args, **kwargs)
187
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
188
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
189
+ [rank0]: return forward_call(*args, **kwargs)
190
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
191
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
192
+ [rank0]: output = func(self, *args, **kwargs)
193
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
194
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
195
+ [rank0]: outputs: BaseModelOutputWithPast = self.model(
196
+ [rank0]: ^^^^^^^^^^^
197
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
198
+ [rank0]: return self._call_impl(*args, **kwargs)
199
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
200
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
201
+ [rank0]: return forward_call(*args, **kwargs)
202
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
203
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
204
+ [rank0]: output = func(self, *args, **kwargs)
205
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
206
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
207
+ [rank0]: outputs = func(self, *args, **kwargs)
208
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
209
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
210
+ [rank0]: hidden_states = decoder_layer(
211
+ [rank0]: ^^^^^^^^^^^^^^
212
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
213
+ [rank0]: return self._call_impl(*args, **kwargs)
214
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
215
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
216
+ [rank0]: return forward_call(*args, **kwargs)
217
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
218
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
219
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
220
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
221
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
222
+ [rank0]: return super().__call__(*args, **kwargs)
223
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
224
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
225
+ [rank0]: return self._call_impl(*args, **kwargs)
226
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
227
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
228
+ [rank0]: return inner()
229
+ [rank0]: ^^^^^^^
230
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
231
+ [rank0]: result = forward_call(*args, **kwargs)
232
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
233
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 332, in forward
234
+ [rank0]: hidden_states = self.mlp(hidden_states)
235
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
236
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
237
+ [rank0]: return self._call_impl(*args, **kwargs)
238
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
239
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
240
+ [rank0]: return forward_call(*args, **kwargs)
241
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
242
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 82, in forward
243
+ [rank0]: down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
244
+ [rank0]: ^^^^^^^^^^^^^^^
245
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
246
+ [rank0]: return self._call_impl(*args, **kwargs)
247
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
248
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
249
+ [rank0]: return forward_call(*args, **kwargs)
250
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
251
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/linear.py", line 134, in forward
252
+ [rank0]: return F.linear(input, self.weight, self.bias)
253
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
254
+ [rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 2.38 GiB. GPU 0 has a total capacity of 139.80 GiB of which 1.40 GiB is free. Including non-PyTorch memory, this process has 138.39 GiB memory in use. Of the allocated memory 135.99 GiB is allocated by PyTorch, and 1.20 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
255
+ wandb:
256
+ wandb: 🚀 View run star-sweep-bs256-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-a9a8c070 at: https://wandb.ai/seyedparsa/thoughtformer/runs/tjrts1mq
257
+ wandb: Find logs at: wandb/run-20260530_033224-tjrts1mq/logs
258
+ E0530 03:32:37.865000 466226 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 466874) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
259
+ Traceback (most recent call last):
260
+ File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
261
+ sys.exit(main())
262
+ ^^^^^^
263
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
264
+ return f(*args, **kwargs)
265
+ ^^^^^^^^^^^^^^^^^^
266
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
267
+ run(args)
268
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
269
+ elastic_launch(
270
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
271
+ return launch_agent(self._config, self._entrypoint, list(args))
272
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
273
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
274
+ raise ChildFailedError(
275
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
276
+ ============================================================
277
+ run.py FAILED
278
+ ------------------------------------------------------------
279
+ Failures:
280
+ <NO_OTHER_FAILURES>
281
+ ------------------------------------------------------------
282
+ Root Cause (first observed failure):
283
+ [0]:
284
+ time : 2026-05-30_03:32:37
285
+ host : ip-172-31-10-226.us-east-2.compute.internal
286
+ rank : 0 (local_rank: 0)
287
+ exitcode : 1 (pid: 466874)
288
+ error_file: <N/A>
289
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
290
+ ============================================================
logs_backup/star_sweep_bs32_lr1e-4.log ADDED
@@ -0,0 +1,152 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs32-lr1e-4', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 32, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 0.0001, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
2
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ return func(*args, **kwargs)
4
+ [rank0]:[W530 03:31:47.851491967 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
5
+
6
+ Running FSDP on rank = 0, world size = 1
7
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
8
+ _init_core_state(
9
+ FullyShardedDataParallel(
10
+ (_fsdp_wrapped_module): Coconut(
11
+ (base_causallm): Qwen3ForCausalLM(
12
+ (model): Qwen3Model(
13
+ (embed_tokens): Embedding(151672, 1024)
14
+ (layers): ModuleList(
15
+ (0-27): 28 x FullyShardedDataParallel(
16
+ (_fsdp_wrapped_module): Qwen3DecoderLayer(
17
+ (self_attn): Qwen3Attention(
18
+ (q_proj): Linear(in_features=1024, out_features=2048, bias=False)
19
+ (k_proj): Linear(in_features=1024, out_features=1024, bias=False)
20
+ (v_proj): Linear(in_features=1024, out_features=1024, bias=False)
21
+ (o_proj): Linear(in_features=2048, out_features=1024, bias=False)
22
+ (q_norm): Qwen3RMSNorm((128,), eps=1e-06)
23
+ (k_norm): Qwen3RMSNorm((128,), eps=1e-06)
24
+ )
25
+ (mlp): Qwen3MLP(
26
+ (gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
27
+ (up_proj): Linear(in_features=1024, out_features=3072, bias=False)
28
+ (down_proj): Linear(in_features=3072, out_features=1024, bias=False)
29
+ (act_fn): SiLUActivation()
30
+ )
31
+ (input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
32
+ (post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
33
+ )
34
+ )
35
+ )
36
+ (norm): Qwen3RMSNorm((1024,), eps=1e-06)
37
+ (rotary_emb): Qwen3RotaryEmbedding()
38
+ )
39
+ (lm_head): Linear(in_features=1024, out_features=151936, bias=False)
40
+ )
41
+ (embedding): Embedding(151672, 1024)
42
+ )
43
+ )
44
+
45
+
46
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
47
+ wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
48
+ wandb: Tracking run with wandb version 0.25.1
49
+ wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033214-gu4dqvfl
50
+ wandb: Run `wandb offline` to turn off syncing.
51
+ wandb: Syncing run star-sweep-bs32-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-d2591099
52
+ wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
53
+ wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/gu4dqvfl
54
+
55
+ ============================================================
56
+ EPOCH 0/30 (thought_stage=1, data_stage=1)
57
+ ============================================================
58
+ thought_stage=1, c_thought=1, max_difficulty=1
59
+
60
+
61
+
62
+
63
+
64
+ LR schedule: cosine, warmup=15 steps, total=9360 steps (max_steps/epoch=312)
65
+
66
+ wandb: WARNING Serializing object of type str that is 419185 bytes
67
+ Traceback (most recent call last):
68
+ File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
69
+ main()
70
+ File "/home/ubuntu/thoughtformer/run.py", line 937, in main
71
+ outputs = parallel_model(**batch)
72
+ ^^^^^^^^^^^^^^^^^^^^^^^
73
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
74
+ return self._call_impl(*args, **kwargs)
75
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
76
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
77
+ return forward_call(*args, **kwargs)
78
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
79
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
80
+ output = self._fsdp_wrapped_module(*args, **kwargs)
81
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
82
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
83
+ return self._call_impl(*args, **kwargs)
84
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
85
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
86
+ return forward_call(*args, **kwargs)
87
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
88
+ File "/home/ubuntu/thoughtformer/coconut.py", line 239, in forward
89
+ logits = torch.cat(logits, dim=-2)
90
+ ^^^^^^^^^^^^^^^^^^^^^^^^^
91
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 14.77 GiB. GPU 0 has a total capacity of 139.80 GiB of which 6.10 GiB is free. Including non-PyTorch memory, this process has 133.69 GiB memory in use. Of the allocated memory 131.74 GiB is allocated by PyTorch, and 767.15 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
92
+ [rank0]: Traceback (most recent call last):
93
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
94
+ [rank0]: main()
95
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
96
+ [rank0]: outputs = parallel_model(**batch)
97
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
98
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
99
+ [rank0]: return self._call_impl(*args, **kwargs)
100
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
101
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
102
+ [rank0]: return forward_call(*args, **kwargs)
103
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
104
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
105
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
106
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
107
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
108
+ [rank0]: return self._call_impl(*args, **kwargs)
109
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
110
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
111
+ [rank0]: return forward_call(*args, **kwargs)
112
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
113
+ [rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 239, in forward
114
+ [rank0]: logits = torch.cat(logits, dim=-2)
115
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^
116
+ [rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 14.77 GiB. GPU 0 has a total capacity of 139.80 GiB of which 6.10 GiB is free. Including non-PyTorch memory, this process has 133.69 GiB memory in use. Of the allocated memory 131.74 GiB is allocated by PyTorch, and 767.15 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
117
+ wandb:
118
+ wandb: 🚀 View run star-sweep-bs32-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-d2591099 at: https://wandb.ai/seyedparsa/thoughtformer/runs/gu4dqvfl
119
+ wandb: Find logs at: wandb/run-20260530_033214-gu4dqvfl/logs
120
+ E0530 03:32:24.440000 464181 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 464254) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
121
+ Traceback (most recent call last):
122
+ File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
123
+ sys.exit(main())
124
+ ^^^^^^
125
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
126
+ return f(*args, **kwargs)
127
+ ^^^^^^^^^^^^^^^^^^
128
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
129
+ run(args)
130
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
131
+ elastic_launch(
132
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
133
+ return launch_agent(self._config, self._entrypoint, list(args))
134
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
135
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
136
+ raise ChildFailedError(
137
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
138
+ ============================================================
139
+ run.py FAILED
140
+ ------------------------------------------------------------
141
+ Failures:
142
+ <NO_OTHER_FAILURES>
143
+ ------------------------------------------------------------
144
+ Root Cause (first observed failure):
145
+ [0]:
146
+ time : 2026-05-30_03:32:24
147
+ host : ip-172-31-10-226.us-east-2.compute.internal
148
+ rank : 0 (local_rank: 0)
149
+ exitcode : 1 (pid: 464254)
150
+ error_file: <N/A>
151
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
152
+ ============================================================
logs_backup/star_sweep_bs32_lr1e-5.log ADDED
@@ -0,0 +1,152 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs32-lr1e-5', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 32, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 1e-05, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
2
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ return func(*args, **kwargs)
4
+ [rank0]:[W530 03:31:45.823267663 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
5
+
6
+ Running FSDP on rank = 0, world size = 1
7
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
8
+ _init_core_state(
9
+ FullyShardedDataParallel(
10
+ (_fsdp_wrapped_module): Coconut(
11
+ (base_causallm): Qwen3ForCausalLM(
12
+ (model): Qwen3Model(
13
+ (embed_tokens): Embedding(151672, 1024)
14
+ (layers): ModuleList(
15
+ (0-27): 28 x FullyShardedDataParallel(
16
+ (_fsdp_wrapped_module): Qwen3DecoderLayer(
17
+ (self_attn): Qwen3Attention(
18
+ (q_proj): Linear(in_features=1024, out_features=2048, bias=False)
19
+ (k_proj): Linear(in_features=1024, out_features=1024, bias=False)
20
+ (v_proj): Linear(in_features=1024, out_features=1024, bias=False)
21
+ (o_proj): Linear(in_features=2048, out_features=1024, bias=False)
22
+ (q_norm): Qwen3RMSNorm((128,), eps=1e-06)
23
+ (k_norm): Qwen3RMSNorm((128,), eps=1e-06)
24
+ )
25
+ (mlp): Qwen3MLP(
26
+ (gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
27
+ (up_proj): Linear(in_features=1024, out_features=3072, bias=False)
28
+ (down_proj): Linear(in_features=3072, out_features=1024, bias=False)
29
+ (act_fn): SiLUActivation()
30
+ )
31
+ (input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
32
+ (post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
33
+ )
34
+ )
35
+ )
36
+ (norm): Qwen3RMSNorm((1024,), eps=1e-06)
37
+ (rotary_emb): Qwen3RotaryEmbedding()
38
+ )
39
+ (lm_head): Linear(in_features=1024, out_features=151936, bias=False)
40
+ )
41
+ (embedding): Embedding(151672, 1024)
42
+ )
43
+ )
44
+
45
+
46
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
47
+ wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
48
+ wandb: Tracking run with wandb version 0.25.1
49
+ wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033212-jek5luxr
50
+ wandb: Run `wandb offline` to turn off syncing.
51
+ wandb: Syncing run star-sweep-bs32-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-c3e91928
52
+ wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
53
+ wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/jek5luxr
54
+
55
+ ============================================================
56
+ EPOCH 0/30 (thought_stage=1, data_stage=1)
57
+ ============================================================
58
+ thought_stage=1, c_thought=1, max_difficulty=1
59
+
60
+
61
+
62
+
63
+
64
+ LR schedule: cosine, warmup=15 steps, total=9360 steps (max_steps/epoch=312)
65
+
66
+ wandb: WARNING Serializing object of type str that is 419185 bytes
67
+ Traceback (most recent call last):
68
+ File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
69
+ main()
70
+ File "/home/ubuntu/thoughtformer/run.py", line 937, in main
71
+ outputs = parallel_model(**batch)
72
+ ^^^^^^^^^^^^^^^^^^^^^^^
73
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
74
+ return self._call_impl(*args, **kwargs)
75
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
76
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
77
+ return forward_call(*args, **kwargs)
78
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
79
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
80
+ output = self._fsdp_wrapped_module(*args, **kwargs)
81
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
82
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
83
+ return self._call_impl(*args, **kwargs)
84
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
85
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
86
+ return forward_call(*args, **kwargs)
87
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
88
+ File "/home/ubuntu/thoughtformer/coconut.py", line 239, in forward
89
+ logits = torch.cat(logits, dim=-2)
90
+ ^^^^^^^^^^^^^^^^^^^^^^^^^
91
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 14.77 GiB. GPU 0 has a total capacity of 139.80 GiB of which 6.10 GiB is free. Including non-PyTorch memory, this process has 133.69 GiB memory in use. Of the allocated memory 131.74 GiB is allocated by PyTorch, and 767.15 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
92
+ [rank0]: Traceback (most recent call last):
93
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
94
+ [rank0]: main()
95
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
96
+ [rank0]: outputs = parallel_model(**batch)
97
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
98
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
99
+ [rank0]: return self._call_impl(*args, **kwargs)
100
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
101
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
102
+ [rank0]: return forward_call(*args, **kwargs)
103
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
104
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
105
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
106
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
107
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
108
+ [rank0]: return self._call_impl(*args, **kwargs)
109
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
110
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
111
+ [rank0]: return forward_call(*args, **kwargs)
112
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
113
+ [rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 239, in forward
114
+ [rank0]: logits = torch.cat(logits, dim=-2)
115
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^
116
+ [rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 14.77 GiB. GPU 0 has a total capacity of 139.80 GiB of which 6.10 GiB is free. Including non-PyTorch memory, this process has 133.69 GiB memory in use. Of the allocated memory 131.74 GiB is allocated by PyTorch, and 767.15 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
117
+ wandb:
118
+ wandb: 🚀 View run star-sweep-bs32-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-c3e91928 at: https://wandb.ai/seyedparsa/thoughtformer/runs/jek5luxr
119
+ wandb: Find logs at: wandb/run-20260530_033212-jek5luxr/logs
120
+ E0530 03:32:23.853000 464041 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 464115) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
121
+ Traceback (most recent call last):
122
+ File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
123
+ sys.exit(main())
124
+ ^^^^^^
125
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
126
+ return f(*args, **kwargs)
127
+ ^^^^^^^^^^^^^^^^^^
128
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
129
+ run(args)
130
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
131
+ elastic_launch(
132
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
133
+ return launch_agent(self._config, self._entrypoint, list(args))
134
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
135
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
136
+ raise ChildFailedError(
137
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
138
+ ============================================================
139
+ run.py FAILED
140
+ ------------------------------------------------------------
141
+ Failures:
142
+ <NO_OTHER_FAILURES>
143
+ ------------------------------------------------------------
144
+ Root Cause (first observed failure):
145
+ [0]:
146
+ time : 2026-05-30_03:32:23
147
+ host : ip-172-31-10-226.us-east-2.compute.internal
148
+ rank : 0 (local_rank: 0)
149
+ exitcode : 1 (pid: 464115)
150
+ error_file: <N/A>
151
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
152
+ ============================================================
logs_backup/star_sweep_bs64_lr1e-4.log ADDED
@@ -0,0 +1,285 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs64-lr1e-4', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 64, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 0.0001, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
2
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ return func(*args, **kwargs)
4
+ [rank0]:[W530 03:31:51.090585317 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
5
+
6
+ Running FSDP on rank = 0, world size = 1
7
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
8
+ _init_core_state(
9
+ FullyShardedDataParallel(
10
+ (_fsdp_wrapped_module): Coconut(
11
+ (base_causallm): Qwen3ForCausalLM(
12
+ (model): Qwen3Model(
13
+ (embed_tokens): Embedding(151672, 1024)
14
+ (layers): ModuleList(
15
+ (0-27): 28 x FullyShardedDataParallel(
16
+ (_fsdp_wrapped_module): Qwen3DecoderLayer(
17
+ (self_attn): Qwen3Attention(
18
+ (q_proj): Linear(in_features=1024, out_features=2048, bias=False)
19
+ (k_proj): Linear(in_features=1024, out_features=1024, bias=False)
20
+ (v_proj): Linear(in_features=1024, out_features=1024, bias=False)
21
+ (o_proj): Linear(in_features=2048, out_features=1024, bias=False)
22
+ (q_norm): Qwen3RMSNorm((128,), eps=1e-06)
23
+ (k_norm): Qwen3RMSNorm((128,), eps=1e-06)
24
+ )
25
+ (mlp): Qwen3MLP(
26
+ (gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
27
+ (up_proj): Linear(in_features=1024, out_features=3072, bias=False)
28
+ (down_proj): Linear(in_features=3072, out_features=1024, bias=False)
29
+ (act_fn): SiLUActivation()
30
+ )
31
+ (input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
32
+ (post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
33
+ )
34
+ )
35
+ )
36
+ (norm): Qwen3RMSNorm((1024,), eps=1e-06)
37
+ (rotary_emb): Qwen3RotaryEmbedding()
38
+ )
39
+ (lm_head): Linear(in_features=1024, out_features=151936, bias=False)
40
+ )
41
+ (embedding): Embedding(151672, 1024)
42
+ )
43
+ )
44
+
45
+
46
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
47
+ wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
48
+ wandb: setting up run edlb2i8x
49
+ wandb: Tracking run with wandb version 0.25.1
50
+ wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033218-edlb2i8x
51
+ wandb: Run `wandb offline` to turn off syncing.
52
+ wandb: Syncing run star-sweep-bs64-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-64cda263
53
+ wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
54
+ wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/edlb2i8x
55
+
56
+ ============================================================
57
+ EPOCH 0/30 (thought_stage=1, data_stage=1)
58
+ ============================================================
59
+ thought_stage=1, c_thought=1, max_difficulty=1
60
+
61
+
62
+
63
+
64
+
65
+ LR schedule: cosine, warmup=7 steps, total=4680 steps (max_steps/epoch=156)
66
+
67
+ wandb: WARNING Serializing object of type str that is 838345 bytes
68
+ Traceback (most recent call last):
69
+ File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
70
+ main()
71
+ File "/home/ubuntu/thoughtformer/run.py", line 937, in main
72
+ outputs = parallel_model(**batch)
73
+ ^^^^^^^^^^^^^^^^^^^^^^^
74
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
75
+ return self._call_impl(*args, **kwargs)
76
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
77
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
78
+ return forward_call(*args, **kwargs)
79
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
80
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
81
+ output = self._fsdp_wrapped_module(*args, **kwargs)
82
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
83
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
84
+ return self._call_impl(*args, **kwargs)
85
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
86
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
87
+ return forward_call(*args, **kwargs)
88
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
89
+ File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
90
+ outputs = self.base_causallm(
91
+ ^^^^^^^^^^^^^^^^^^^
92
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
93
+ return self._call_impl(*args, **kwargs)
94
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
95
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
96
+ return forward_call(*args, **kwargs)
97
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
98
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
99
+ output = func(self, *args, **kwargs)
100
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
101
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
102
+ outputs: BaseModelOutputWithPast = self.model(
103
+ ^^^^^^^^^^^
104
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
105
+ return self._call_impl(*args, **kwargs)
106
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
107
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
108
+ return forward_call(*args, **kwargs)
109
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
110
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
111
+ output = func(self, *args, **kwargs)
112
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
113
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
114
+ outputs = func(self, *args, **kwargs)
115
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
116
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
117
+ hidden_states = decoder_layer(
118
+ ^^^^^^^^^^^^^^
119
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
120
+ return self._call_impl(*args, **kwargs)
121
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
122
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
123
+ return forward_call(*args, **kwargs)
124
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
125
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
126
+ output = self._fsdp_wrapped_module(*args, **kwargs)
127
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
128
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
129
+ return super().__call__(*args, **kwargs)
130
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
131
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
132
+ return self._call_impl(*args, **kwargs)
133
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
134
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
135
+ return inner()
136
+ ^^^^^^^
137
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
138
+ result = forward_call(*args, **kwargs)
139
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
140
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 318, in forward
141
+ hidden_states, _ = self.self_attn(
142
+ ^^^^^^^^^^^^^^^
143
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
144
+ return self._call_impl(*args, **kwargs)
145
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
146
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
147
+ return inner()
148
+ ^^^^^^^
149
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
150
+ result = forward_call(*args, **kwargs)
151
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
152
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 277, in forward
153
+ attn_output, attn_weights = attention_interface(
154
+ ^^^^^^^^^^^^^^^^^^^^
155
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/integrations/sdpa_attention.py", line 92, in sdpa_attention_forward
156
+ attn_output = torch.nn.functional.scaled_dot_product_attention(
157
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
158
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 406.00 MiB. GPU 0 has a total capacity of 139.80 GiB of which 45.06 MiB is free. Including non-PyTorch memory, this process has 139.75 GiB memory in use. Of the allocated memory 137.33 GiB is allocated by PyTorch, and 1.22 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
159
+ [rank0]: Traceback (most recent call last):
160
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
161
+ [rank0]: main()
162
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
163
+ [rank0]: outputs = parallel_model(**batch)
164
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
165
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
166
+ [rank0]: return self._call_impl(*args, **kwargs)
167
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
168
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
169
+ [rank0]: return forward_call(*args, **kwargs)
170
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
171
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
172
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
173
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
174
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
175
+ [rank0]: return self._call_impl(*args, **kwargs)
176
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
177
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
178
+ [rank0]: return forward_call(*args, **kwargs)
179
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
180
+ [rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
181
+ [rank0]: outputs = self.base_causallm(
182
+ [rank0]: ^^^^^^^^^^^^^^^^^^^
183
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
184
+ [rank0]: return self._call_impl(*args, **kwargs)
185
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
186
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
187
+ [rank0]: return forward_call(*args, **kwargs)
188
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
189
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
190
+ [rank0]: output = func(self, *args, **kwargs)
191
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
192
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
193
+ [rank0]: outputs: BaseModelOutputWithPast = self.model(
194
+ [rank0]: ^^^^^^^^^^^
195
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
196
+ [rank0]: return self._call_impl(*args, **kwargs)
197
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
198
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
199
+ [rank0]: return forward_call(*args, **kwargs)
200
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
201
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
202
+ [rank0]: output = func(self, *args, **kwargs)
203
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
204
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
205
+ [rank0]: outputs = func(self, *args, **kwargs)
206
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
207
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
208
+ [rank0]: hidden_states = decoder_layer(
209
+ [rank0]: ^^^^^^^^^^^^^^
210
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
211
+ [rank0]: return self._call_impl(*args, **kwargs)
212
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
213
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
214
+ [rank0]: return forward_call(*args, **kwargs)
215
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
216
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
217
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
218
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
219
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
220
+ [rank0]: return super().__call__(*args, **kwargs)
221
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
222
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
223
+ [rank0]: return self._call_impl(*args, **kwargs)
224
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
225
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
226
+ [rank0]: return inner()
227
+ [rank0]: ^^^^^^^
228
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
229
+ [rank0]: result = forward_call(*args, **kwargs)
230
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
231
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 318, in forward
232
+ [rank0]: hidden_states, _ = self.self_attn(
233
+ [rank0]: ^^^^^^^^^^^^^^^
234
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
235
+ [rank0]: return self._call_impl(*args, **kwargs)
236
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
237
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
238
+ [rank0]: return inner()
239
+ [rank0]: ^^^^^^^
240
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
241
+ [rank0]: result = forward_call(*args, **kwargs)
242
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
243
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 277, in forward
244
+ [rank0]: attn_output, attn_weights = attention_interface(
245
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^
246
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/integrations/sdpa_attention.py", line 92, in sdpa_attention_forward
247
+ [rank0]: attn_output = torch.nn.functional.scaled_dot_product_attention(
248
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
249
+ [rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 406.00 MiB. GPU 0 has a total capacity of 139.80 GiB of which 45.06 MiB is free. Including non-PyTorch memory, this process has 139.75 GiB memory in use. Of the allocated memory 137.33 GiB is allocated by PyTorch, and 1.22 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
250
+ wandb:
251
+ wandb: 🚀 View run star-sweep-bs64-lr1e-4_n1000_qwen3-0.6b-base_thoughtformer_lr1e-4_no-reset_dc_tc_adaptive-64cda263 at: https://wandb.ai/seyedparsa/thoughtformer/runs/edlb2i8x
252
+ wandb: Find logs at: wandb/run-20260530_033218-edlb2i8x/logs
253
+ E0530 03:32:29.958000 464466 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 464939) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
254
+ Traceback (most recent call last):
255
+ File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
256
+ sys.exit(main())
257
+ ^^^^^^
258
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
259
+ return f(*args, **kwargs)
260
+ ^^^^^^^^^^^^^^^^^^
261
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
262
+ run(args)
263
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
264
+ elastic_launch(
265
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
266
+ return launch_agent(self._config, self._entrypoint, list(args))
267
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
268
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
269
+ raise ChildFailedError(
270
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
271
+ ============================================================
272
+ run.py FAILED
273
+ ------------------------------------------------------------
274
+ Failures:
275
+ <NO_OTHER_FAILURES>
276
+ ------------------------------------------------------------
277
+ Root Cause (first observed failure):
278
+ [0]:
279
+ time : 2026-05-30_03:32:29
280
+ host : ip-172-31-10-226.us-east-2.compute.internal
281
+ rank : 0 (local_rank: 0)
282
+ exitcode : 1 (pid: 464939)
283
+ error_file: <N/A>
284
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
285
+ ============================================================
logs_backup/star_sweep_bs64_lr1e-5.log ADDED
@@ -0,0 +1,284 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Config: {'project': 'thoughtformer', 'name': 'star-sweep-bs64-lr1e-5', 'only_eval': False, 'method': 'thoughtformer', 'data_curriculum': True, 'thought_curriculum': True, 'staging': 'adaptive', 'init_thought_stage': 1, 'init_data_stage': 1, 'patience': 5, 'staging_threshold': 0.9, 'c_thought': 1, 'max_latent_stage': 10, 'uniform_prob': 0.1, 'save_only_improve': False, 'model_id': 'Qwen/Qwen3-0.6B-Base', 'load_model_path': 'None', 'seed': 0, 'resume': 0, 'bf16': False, 'train_path': 'data/star_k10_L10_1000_train.json', 'val_path': 'data/star_k10_L10_50_valid.json', 'reset_optimizer': False, 'lr_schedule': 'cosine', 'batch_size_training': 64, 'eval_only_trained': False, 'eval_batch_size': 8, 'eval_every': 1, 'debug': False, 'gradient_accumulation_steps': 1, 'num_epochs': 30, 'lr': 1e-05, 'weight_decay': 0.01, 'group': 'star_bs_lr_sweep', 'tags': ['star', 'k10', 'L10', 'atc', 'adaptive', 'lr1e-5', 'best', 'budget_aug'], 'run_type': 'pilot'}
2
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ return func(*args, **kwargs)
4
+ [rank0]:[W530 03:31:49.899347241 ProcessGroupNCCL.cpp:5324] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
5
+
6
+ Running FSDP on rank = 0, world size = 1
7
+ /home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py:478: UserWarning: FSDP is switching to use `NO_SHARD` instead of ShardingStrategy.FULL_SHARD since the world size is 1.
8
+ _init_core_state(
9
+ FullyShardedDataParallel(
10
+ (_fsdp_wrapped_module): Coconut(
11
+ (base_causallm): Qwen3ForCausalLM(
12
+ (model): Qwen3Model(
13
+ (embed_tokens): Embedding(151672, 1024)
14
+ (layers): ModuleList(
15
+ (0-27): 28 x FullyShardedDataParallel(
16
+ (_fsdp_wrapped_module): Qwen3DecoderLayer(
17
+ (self_attn): Qwen3Attention(
18
+ (q_proj): Linear(in_features=1024, out_features=2048, bias=False)
19
+ (k_proj): Linear(in_features=1024, out_features=1024, bias=False)
20
+ (v_proj): Linear(in_features=1024, out_features=1024, bias=False)
21
+ (o_proj): Linear(in_features=2048, out_features=1024, bias=False)
22
+ (q_norm): Qwen3RMSNorm((128,), eps=1e-06)
23
+ (k_norm): Qwen3RMSNorm((128,), eps=1e-06)
24
+ )
25
+ (mlp): Qwen3MLP(
26
+ (gate_proj): Linear(in_features=1024, out_features=3072, bias=False)
27
+ (up_proj): Linear(in_features=1024, out_features=3072, bias=False)
28
+ (down_proj): Linear(in_features=3072, out_features=1024, bias=False)
29
+ (act_fn): SiLUActivation()
30
+ )
31
+ (input_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
32
+ (post_attention_layernorm): Qwen3RMSNorm((1024,), eps=1e-06)
33
+ )
34
+ )
35
+ )
36
+ (norm): Qwen3RMSNorm((1024,), eps=1e-06)
37
+ (rotary_emb): Qwen3RotaryEmbedding()
38
+ )
39
+ (lm_head): Linear(in_features=1024, out_features=151936, bias=False)
40
+ )
41
+ (embedding): Embedding(151672, 1024)
42
+ )
43
+ )
44
+
45
+
46
+ wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from /home/ubuntu/.netrc.
47
+ wandb: Currently logged in as: seyedparsa to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
48
+ wandb: Tracking run with wandb version 0.25.1
49
+ wandb: Run data is saved locally in /home/ubuntu/thoughtformer/wandb/run-20260530_033216-nwpqaes3
50
+ wandb: Run `wandb offline` to turn off syncing.
51
+ wandb: Syncing run star-sweep-bs64-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-c9ba8dcb
52
+ wandb: ⭐️ View project at https://wandb.ai/seyedparsa/thoughtformer
53
+ wandb: 🚀 View run at https://wandb.ai/seyedparsa/thoughtformer/runs/nwpqaes3
54
+
55
+ ============================================================
56
+ EPOCH 0/30 (thought_stage=1, data_stage=1)
57
+ ============================================================
58
+ thought_stage=1, c_thought=1, max_difficulty=1
59
+
60
+
61
+
62
+
63
+
64
+ LR schedule: cosine, warmup=7 steps, total=4680 steps (max_steps/epoch=156)
65
+
66
+ wandb: WARNING Serializing object of type str that is 838345 bytes
67
+ Traceback (most recent call last):
68
+ File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
69
+ main()
70
+ File "/home/ubuntu/thoughtformer/run.py", line 937, in main
71
+ outputs = parallel_model(**batch)
72
+ ^^^^^^^^^^^^^^^^^^^^^^^
73
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
74
+ return self._call_impl(*args, **kwargs)
75
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
76
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
77
+ return forward_call(*args, **kwargs)
78
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
79
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
80
+ output = self._fsdp_wrapped_module(*args, **kwargs)
81
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
82
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
83
+ return self._call_impl(*args, **kwargs)
84
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
85
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
86
+ return forward_call(*args, **kwargs)
87
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
88
+ File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
89
+ outputs = self.base_causallm(
90
+ ^^^^^^^^^^^^^^^^^^^
91
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
92
+ return self._call_impl(*args, **kwargs)
93
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
94
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
95
+ return forward_call(*args, **kwargs)
96
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
97
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
98
+ output = func(self, *args, **kwargs)
99
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
100
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
101
+ outputs: BaseModelOutputWithPast = self.model(
102
+ ^^^^^^^^^^^
103
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
104
+ return self._call_impl(*args, **kwargs)
105
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
106
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
107
+ return forward_call(*args, **kwargs)
108
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
109
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
110
+ output = func(self, *args, **kwargs)
111
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
112
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
113
+ outputs = func(self, *args, **kwargs)
114
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
115
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
116
+ hidden_states = decoder_layer(
117
+ ^^^^^^^^^^^^^^
118
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
119
+ return self._call_impl(*args, **kwargs)
120
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
121
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
122
+ return forward_call(*args, **kwargs)
123
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
124
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
125
+ output = self._fsdp_wrapped_module(*args, **kwargs)
126
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
127
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
128
+ return super().__call__(*args, **kwargs)
129
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
130
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
131
+ return self._call_impl(*args, **kwargs)
132
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
133
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
134
+ return inner()
135
+ ^^^^^^^
136
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
137
+ result = forward_call(*args, **kwargs)
138
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
139
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 318, in forward
140
+ hidden_states, _ = self.self_attn(
141
+ ^^^^^^^^^^^^^^^
142
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
143
+ return self._call_impl(*args, **kwargs)
144
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
145
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
146
+ return inner()
147
+ ^^^^^^^
148
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
149
+ result = forward_call(*args, **kwargs)
150
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
151
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 277, in forward
152
+ attn_output, attn_weights = attention_interface(
153
+ ^^^^^^^^^^^^^^^^^^^^
154
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/integrations/sdpa_attention.py", line 92, in sdpa_attention_forward
155
+ attn_output = torch.nn.functional.scaled_dot_product_attention(
156
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
157
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 406.00 MiB. GPU 0 has a total capacity of 139.80 GiB of which 45.06 MiB is free. Including non-PyTorch memory, this process has 139.75 GiB memory in use. Of the allocated memory 137.33 GiB is allocated by PyTorch, and 1.22 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
158
+ [rank0]: Traceback (most recent call last):
159
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 1194, in <module>
160
+ [rank0]: main()
161
+ [rank0]: File "/home/ubuntu/thoughtformer/run.py", line 937, in main
162
+ [rank0]: outputs = parallel_model(**batch)
163
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^
164
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
165
+ [rank0]: return self._call_impl(*args, **kwargs)
166
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
167
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
168
+ [rank0]: return forward_call(*args, **kwargs)
169
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
170
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
171
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
172
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
173
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
174
+ [rank0]: return self._call_impl(*args, **kwargs)
175
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
176
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
177
+ [rank0]: return forward_call(*args, **kwargs)
178
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
179
+ [rank0]: File "/home/ubuntu/thoughtformer/coconut.py", line 101, in forward
180
+ [rank0]: outputs = self.base_causallm(
181
+ [rank0]: ^^^^^^^^^^^^^^^^^^^
182
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
183
+ [rank0]: return self._call_impl(*args, **kwargs)
184
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
185
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
186
+ [rank0]: return forward_call(*args, **kwargs)
187
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
188
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 857, in wrapper
189
+ [rank0]: output = func(self, *args, **kwargs)
190
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
191
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 492, in forward
192
+ [rank0]: outputs: BaseModelOutputWithPast = self.model(
193
+ [rank0]: ^^^^^^^^^^^
194
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
195
+ [rank0]: return self._call_impl(*args, **kwargs)
196
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
197
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
198
+ [rank0]: return forward_call(*args, **kwargs)
199
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
200
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/generic.py", line 931, in wrapper
201
+ [rank0]: output = func(self, *args, **kwargs)
202
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
203
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/utils/output_capturing.py", line 248, in wrapper
204
+ [rank0]: outputs = func(self, *args, **kwargs)
205
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
206
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 424, in forward
207
+ [rank0]: hidden_states = decoder_layer(
208
+ [rank0]: ^^^^^^^^^^^^^^
209
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
210
+ [rank0]: return self._call_impl(*args, **kwargs)
211
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
212
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl
213
+ [rank0]: return forward_call(*args, **kwargs)
214
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
215
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/fsdp/fully_sharded_data_parallel.py", line 857, in forward
216
+ [rank0]: output = self._fsdp_wrapped_module(*args, **kwargs)
217
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
218
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/modeling_layers.py", line 93, in __call__
219
+ [rank0]: return super().__call__(*args, **kwargs)
220
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
221
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
222
+ [rank0]: return self._call_impl(*args, **kwargs)
223
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
224
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
225
+ [rank0]: return inner()
226
+ [rank0]: ^^^^^^^
227
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
228
+ [rank0]: result = forward_call(*args, **kwargs)
229
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
230
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 318, in forward
231
+ [rank0]: hidden_states, _ = self.self_attn(
232
+ [rank0]: ^^^^^^^^^^^^^^^
233
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl
234
+ [rank0]: return self._call_impl(*args, **kwargs)
235
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
236
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1884, in _call_impl
237
+ [rank0]: return inner()
238
+ [rank0]: ^^^^^^^
239
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1832, in inner
240
+ [rank0]: result = forward_call(*args, **kwargs)
241
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
242
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/models/qwen3/modeling_qwen3.py", line 277, in forward
243
+ [rank0]: attn_output, attn_weights = attention_interface(
244
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^
245
+ [rank0]: File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/transformers/integrations/sdpa_attention.py", line 92, in sdpa_attention_forward
246
+ [rank0]: attn_output = torch.nn.functional.scaled_dot_product_attention(
247
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
248
+ [rank0]: torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 406.00 MiB. GPU 0 has a total capacity of 139.80 GiB of which 45.06 MiB is free. Including non-PyTorch memory, this process has 139.75 GiB memory in use. Of the allocated memory 137.33 GiB is allocated by PyTorch, and 1.22 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
249
+ wandb:
250
+ wandb: 🚀 View run star-sweep-bs64-lr1e-5_n1000_qwen3-0.6b-base_thoughtformer_lr1e-5_no-reset_dc_tc_adaptive-c9ba8dcb at: https://wandb.ai/seyedparsa/thoughtformer/runs/nwpqaes3
251
+ wandb: Find logs at: wandb/run-20260530_033216-nwpqaes3/logs
252
+ E0530 03:32:27.283000 464320 torch/distributed/elastic/multiprocessing/api.py:988] failed (exitcode: 1) local_rank: 0 (pid: 464399) of binary: /home/ubuntu/thoughtformer/.venv/bin/python3
253
+ Traceback (most recent call last):
254
+ File "/home/ubuntu/thoughtformer/.venv/bin/torchrun", line 8, in <module>
255
+ sys.exit(main())
256
+ ^^^^^^
257
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/elastic/multiprocessing/errors/__init__.py", line 367, in wrapper
258
+ return f(*args, **kwargs)
259
+ ^^^^^^^^^^^^^^^^^^
260
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1016, in main
261
+ run(args)
262
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/run.py", line 1007, in run
263
+ elastic_launch(
264
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 191, in __call__
265
+ return launch_agent(self._config, self._entrypoint, list(args))
266
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
267
+ File "/home/ubuntu/thoughtformer/.venv/lib/python3.12/site-packages/torch/distributed/launcher/api.py", line 371, in launch_agent
268
+ raise ChildFailedError(
269
+ torch.distributed.elastic.multiprocessing.errors.ChildFailedError:
270
+ ============================================================
271
+ run.py FAILED
272
+ ------------------------------------------------------------
273
+ Failures:
274
+ <NO_OTHER_FAILURES>
275
+ ------------------------------------------------------------
276
+ Root Cause (first observed failure):
277
+ [0]:
278
+ time : 2026-05-30_03:32:27
279
+ host : ip-172-31-10-226.us-east-2.compute.internal
280
+ rank : 0 (local_rank: 0)
281
+ exitcode : 1 (pid: 464399)
282
+ error_file: <N/A>
283
+ traceback : To enable traceback see: https://pytorch.org/docs/stable/elastic/errors.html
284
+ ============================================================
logs_backup/star_thoughtformer.log ADDED
The diff for this file is too large to render. See raw diff