diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_000500.json b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_000500.json deleted file mode 100644 index ba79b4dc95de94cd09f87814f615475ece474c1e..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_000500.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 500, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.5891939434241213, - "total_training_time": 1312.64857006073 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_001000.json b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_001000.json deleted file mode 100644 index bbca0e8342d8f36e4a06b1933ce176471b508a07..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_001000.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 1000, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.4868806829553485, - "total_training_time": 2654.1426842212677 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_001500.json b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_001500.json deleted file mode 100644 index 4d052d841af639e6ed386c3bf952cdef76107cbf..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_001500.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 1500, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.410999880107203, - "total_training_time": 3995.465323448181 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_002000.json b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_002000.json deleted file mode 100644 index 8ccecb680019f8c6b8dd9c312d17235d307c95bf..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_002000.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 2000, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.452807694528099, - "total_training_time": 5336.5537366867065 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_002500.json b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_002500.json deleted file mode 100644 index 96fc7e712e052057bcc504c98aa1891ccee8b9e2..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_002500.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 2500, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 13, - "pos": 10792769, - "epoch": 1, - "pq_idx": 13, - "rg_idx": 10792769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.327396609246394, - "total_training_time": 6677.6987290382385 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_003000.json b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_003000.json deleted file mode 100644 index 7ea0f8f11e2301baef3d24fa3c60d0fa0a52f9de..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_003000.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 3000, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 15, - "pos": 72944769, - "epoch": 1, - "pq_idx": 15, - "rg_idx": 72944769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.048554485031434, - "total_training_time": 8018.697687149048 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_003500.json b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_003500.json deleted file mode 100644 index ea7c74460fdec7160133b9cbc32e9e20470153ee..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_003500.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 3500, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 18, - "pos": 35096769, - "epoch": 1, - "pq_idx": 18, - "rg_idx": 35096769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.0863692227626416, - "total_training_time": 9359.67183303833 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_004000.json b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_004000.json deleted file mode 100644 index e43944a04fee67ab2d03020bbd670e2b024bf741..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_004000.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 4000, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 20, - "pos": 97248769, - "epoch": 1, - "pq_idx": 20, - "rg_idx": 97248769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 2.7993042062101186, - "total_training_time": 10700.389838218689 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_004200.json b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_004200.json deleted file mode 100644 index edcf82bc7f5d7ab29272189b9203fe91c518bbe2..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/meta_004200.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 4200, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 22, - "pos": 2109569, - "epoch": 1, - "pq_idx": 22, - "rg_idx": 2109569 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 2.8461822660242255, - "total_training_time": 11236.57096004486 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_000500.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_000500.pt deleted file mode 100644 index 97e4c71cc3328cef1fabf54e4a1b8ecf3e240e32..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:00e14e7ad4637b08c6e5f649ebc7de7ed0d9fcb39a293b7b088a13c2ebb549f8 -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_001000.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_001000.pt deleted file mode 100644 index e2d60950935b1958573cb2de2aa7a12151bc1f16..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:01e2a9466841de94c8c3a654a257c842c1764d3ac4cd79cf478f9f2b24859e7c -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_001500.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_001500.pt deleted file mode 100644 index 5bad12b7a4a0a3593d00a4ba4d742ef0bd7e5945..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:03fb5d4241142f86b69396fe200295929df1b4bcc2ba5d4cc464c9627cf0a530 -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_002000.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_002000.pt deleted file mode 100644 index 283db3b4a5e606111e288f51ab830fceb5ea3fb3..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c4db8d9fa2b5fa4ca14f65dfd8d5356c691c5c8772e839555f0fec6f8d300269 -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_002500.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_002500.pt deleted file mode 100644 index 766aefe9911a42ba0478ae0effee731d7e248fe1..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:abb54590a71621be44e418b133a42088aed1d84872a9726e9aa4ac8b7e418788 -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_003000.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_003000.pt deleted file mode 100644 index 95408589ebb4a00a04f89c866519d954a17a96db..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_003000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0a96aa9105d455f741602d43d807dd0b2732823c9c8be3465717f7090aa469eb -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_003500.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_003500.pt deleted file mode 100644 index aa7696886c6993d18f98632b2a12d2e93c89c890..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_003500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bab4a0741306ce5945c9a86dabd233081b836f0e1f814dd1d9363cc1b9cce950 -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_004000.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_004000.pt deleted file mode 100644 index 195c00c1e8c9542b4c3a4237f783261f127da27b..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_004000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6c967e927215ebca6af3f49011d98fd8fa4487c55e78485271878e382b7293d5 -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_004200.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_004200.pt deleted file mode 100644 index c201085206abc74184928a0dac513e531432f6cc..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/model_004200.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1dfab1c19f9bbd3dd27ed859b70cfd242d5ebaef734985dbef78d4998493a6f6 -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_000500_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_000500_rank0.pt deleted file mode 100644 index 3c6f2f796a2895cf54026c01b22c14a52177d24e..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:62b942e737b779c2dbb3b14246ffc951fb9dce816e10e7ada936b81093faaca7 -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_001000_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_001000_rank0.pt deleted file mode 100644 index 0272991a5a4536ff457d456339de4d750c8cda24..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f5fe23783669dcb79dda4bc0a66e2c536d9d3055fbcaf07840662e2c5b8f6950 -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_001500_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_001500_rank0.pt deleted file mode 100644 index 88a0adfe1b99ce52dcf0807bf3881371f4ed1184..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9dcce76cabc6cdd8083af5f064ae391d8b84fc96d2608ab6740fffc2b1e2f5f2 -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_002000_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_002000_rank0.pt deleted file mode 100644 index b11fbe743399c95fabbd0696c2852a595104d922..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4dc97c55d9aa7e22cd38dcfc298b42066aeda741daf7d1b508575ac62c0e32de -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_002500_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_002500_rank0.pt deleted file mode 100644 index 37d285b746991053e40d3e97b32448203c24c826..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2550badb9d208d9f2e51765bafd6789cc2f9ee8b6fd5880e25790d427deb9556 -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_003000_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_003000_rank0.pt deleted file mode 100644 index b78dee7e866f9e7c539a9d6ca1e48264d7b2c2b5..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_003000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8ea5ec19b0375980f64cbdbb9709d040a01e2f546ebdcc380246f8c69d26b242 -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_003500_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_003500_rank0.pt deleted file mode 100644 index 26f63d83d8e08a5bf8c6b3139f798f5dabfb10aa..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_003500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9465372371e5d248972fe855b83234cbf5fde5bcf208eb7e2faf9f51cf207def -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_004000_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_004000_rank0.pt deleted file mode 100644 index 8ad1e5577fd1f8636688a15b9f932609943770e6..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_004000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c27aa4c1fea51a61cd5aef3654780852040f59b347b9f18c7a9f29620664b677 -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_004200_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_004200_rank0.pt deleted file mode 100644 index dcfacec3437fca29f63311b48f8f86a813607bcd..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12-ratio20/optim_004200_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3a71d1f755bf5e13a1bd7741a3a1bbc1d9807468b5190c4e2380b91e0116eee0 -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/meta_000500.json b/archive/pre-lineage-v1/base_checkpoints/d12/meta_000500.json deleted file mode 100644 index 4346b0a0961c0461b449f5bf3e53e0e367f9f980..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/meta_000500.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 500, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": null - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.5893779623775406, - "total_training_time": 1290.4532148838043 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/meta_001000.json b/archive/pre-lineage-v1/base_checkpoints/d12/meta_001000.json deleted file mode 100644 index d44aff60900e63f44ee57a352b66725c582f2c36..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/meta_001000.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 1000, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": null - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.4419165825253133, - "total_training_time": 2609.577807664871 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/meta_001500.json b/archive/pre-lineage-v1/base_checkpoints/d12/meta_001500.json deleted file mode 100644 index 1991d9750741add910d01954f8482d4e2247e360..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/meta_001500.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 1500, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": null - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.2465426140603015, - "total_training_time": 3929.598204135895 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/meta_002000.json b/archive/pre-lineage-v1/base_checkpoints/d12/meta_002000.json deleted file mode 100644 index 488a6f07b4fea050f2e89f34893aad59564826a0..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/meta_002000.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 2000, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": null - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.2442641345309373, - "total_training_time": 5249.494728565216 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/meta_002362.json b/archive/pre-lineage-v1/base_checkpoints/d12/meta_002362.json deleted file mode 100644 index e175df81e356e808f571537ad3eb29ed517c3335..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/meta_002362.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 2362, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": null - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.074421420856799, - "total_training_time": 6205.646646976471 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/model_000500.pt b/archive/pre-lineage-v1/base_checkpoints/d12/model_000500.pt deleted file mode 100644 index adb2ea235f50e379dec7aedafe47be61861bedab..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:21fbdff366e5db3fa2f254b4b98c763cc70ba95722242fa032d8d21b956b694f -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/model_001000.pt b/archive/pre-lineage-v1/base_checkpoints/d12/model_001000.pt deleted file mode 100644 index 6c50be693db2de57ed0fc9edeb4808310b2631ab..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6f0a862044e51b8cac250cf1e4f26af7f8347fd887b8c30af08addb879b6494d -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/model_001500.pt b/archive/pre-lineage-v1/base_checkpoints/d12/model_001500.pt deleted file mode 100644 index 6c11f7b0d9217759f1212acf26a448949fffd5d8..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:326bddc26c100de33310f9164dc873af6099a9b7760527a8707c256988a8ec7a -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/model_002000.pt b/archive/pre-lineage-v1/base_checkpoints/d12/model_002000.pt deleted file mode 100644 index 4ec265330fa790437406e93a5448ad62eb39fe99..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a5d101975722cffc43eb4f35aa967a5afd397b159da3521e5a1acbd3819d466e -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/model_002362.pt b/archive/pre-lineage-v1/base_checkpoints/d12/model_002362.pt deleted file mode 100644 index 6cf06b066423745b22aa44519e4b864d5bdc27d6..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:04331c52ed8fa7259f350e4ec72d0dd6602451cfd75a5773a4c17ac5c141ea7e -size 792761399 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/optim_000500_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12/optim_000500_rank0.pt deleted file mode 100644 index 79c1822fb2dc78fb93423cdc8813dfc59fea2a94..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d39a131089a476a202ae932f21d9398b225d0b8e4147a58fbdff797914d34976 -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/optim_001000_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12/optim_001000_rank0.pt deleted file mode 100644 index f055caef01d26659f49ac7ae7f17083f45d82d4a..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e4971566eb3d4aa5d5fe29b3a3b77f56581b61713e5c1d162debb8a409c02118 -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/optim_001500_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12/optim_001500_rank0.pt deleted file mode 100644 index 275f823331fcb1cd614bc3d8e15f97265d98d85a..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:05ce87d44bfd6e9b4d4c1e643a2a5aa4b169119913ccbdf3625d7cfa7313b342 -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/optim_002000_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12/optim_002000_rank0.pt deleted file mode 100644 index b3b5e1bd94301a192b14e234e17f4f28786b5538..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1b58f68fd887360ce99e8456aecf706b1bcf5c8210194721389698ec658237b1 -size 1246165237 diff --git a/archive/pre-lineage-v1/base_checkpoints/d12/optim_002362_rank0.pt b/archive/pre-lineage-v1/base_checkpoints/d12/optim_002362_rank0.pt deleted file mode 100644 index 47439d09d2d9170f2458f5ad529c461a8e8c7f80..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/base_checkpoints/d12/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:09aa5632a4b33981a1b2fbd97254d0050ecb7916d0c242f1d195c0726be314d7 -size 1246165237 diff --git a/archive/pre-lineage-v1/chatsft_checkpoints/d12/meta_001065.json b/archive/pre-lineage-v1/chatsft_checkpoints/d12/meta_001065.json deleted file mode 100644 index 2f52697c9b62b9b51836bb31924aadd0d871459c..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/chatsft_checkpoints/d12/meta_001065.json +++ /dev/null @@ -1,38 +0,0 @@ -{ - "step": 1065, - "val_bpb": 0.39271438585109175, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "model_tag": "d12", - "model_step": null, - "load_optimizer": 1, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.0, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": -1, - "eval_tokens": 20971520, - "chatcore_every": -1, - "chatcore_max_cat": -1, - "chatcore_max_sample": 24, - "mmlu_epochs": 3, - "gsm8k_epochs": 4 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/chatsft_checkpoints/d12/model_001065.pt b/archive/pre-lineage-v1/chatsft_checkpoints/d12/model_001065.pt deleted file mode 100644 index 637ffa7aa73462ac49b2503b5b34b31bfcd7b3ce..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/chatsft_checkpoints/d12/model_001065.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f1ef8886ea2cbd820baa9bc98368f99673efb1f5fdbd082196d573b291111bda -size 792761399 diff --git a/archive/pre-lineage-v1/chatsft_checkpoints/d12/optim_001065_rank0.pt b/archive/pre-lineage-v1/chatsft_checkpoints/d12/optim_001065_rank0.pt deleted file mode 100644 index 451dddf0789f3113d3b31c3792c2328b1f92da2c..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/chatsft_checkpoints/d12/optim_001065_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:db8c9df6e288618ed6362102e29edc3731267b4ff382154a55709b4d5a14f1d4 -size 1246165237 diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/meta_000500.json b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/meta_000500.json deleted file mode 100644 index 8327e3a6e8e9c33ea647dc4e340437666e3cb2e4..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,125 +0,0 @@ -{ - "step": 500, - "experiment_id": "climbmix-d12-r12-baseline-v2", - "val_bpb": 1.039594309585361, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "climbmix-d12-r12-baseline-v2", - "wandb_run_id": "412c294c", - "wandb_group": "climbmix-d12", - "wandb_tags": "climbmix,d12,baseline,ratio12", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints", - "experiment_id": "climbmix-d12-r12-baseline-v2", - "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/config.json", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "climbmix-d12-r12-baseline-v2", - "experiment": { - "experiment_id": "climbmix-d12-r12-baseline-v2", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 8, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "storage": { - "hf_model_repo": "jbduran/think-nanochat-d12", - "path": "experiments/climbmix-d12-r12-baseline-v2" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "climbmix-d12-r12-baseline-v2", - "group": "climbmix-d12", - "tags": [ - "climbmix", - "d12", - "baseline", - "ratio12" - ] - } - } - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.039594309585361, - "smooth_train_loss": 3.4405327881291092, - "total_training_time": 1309.0715219974518 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/meta_001000.json b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/meta_001000.json deleted file mode 100644 index 05ed68f23ea40581b4214349adae7df687e5adc8..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,125 +0,0 @@ -{ - "step": 1000, - "experiment_id": "climbmix-d12-r12-baseline-v2", - "val_bpb": 0.9812561487708248, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "climbmix-d12-r12-baseline-v2", - "wandb_run_id": "412c294c", - "wandb_group": "climbmix-d12", - "wandb_tags": "climbmix,d12,baseline,ratio12", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints", - "experiment_id": "climbmix-d12-r12-baseline-v2", - "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/config.json", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "climbmix-d12-r12-baseline-v2", - "experiment": { - "experiment_id": "climbmix-d12-r12-baseline-v2", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 8, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "storage": { - "hf_model_repo": "jbduran/think-nanochat-d12", - "path": "experiments/climbmix-d12-r12-baseline-v2" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "climbmix-d12-r12-baseline-v2", - "group": "climbmix-d12", - "tags": [ - "climbmix", - "d12", - "baseline", - "ratio12" - ] - } - } - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 0, - "pos": 93414403, - "epoch": 2, - "pq_idx": 0, - "rg_idx": 93414403 - }, - "loop_state": { - "min_val_bpb": 0.9812561487708248, - "smooth_train_loss": 3.1852821933493254, - "total_training_time": 2645.4519686698914 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/meta_001500.json b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/meta_001500.json deleted file mode 100644 index 2f73bed6c5dfd40872806339e262208d8ba5e040..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,125 +0,0 @@ -{ - "step": 1500, - "experiment_id": "climbmix-d12-r12-baseline-v2", - "val_bpb": 0.9399637061635419, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "climbmix-d12-r12-baseline-v2", - "wandb_run_id": "412c294c", - "wandb_group": "climbmix-d12", - "wandb_tags": "climbmix,d12,baseline,ratio12", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints", - "experiment_id": "climbmix-d12-r12-baseline-v2", - "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-r12-baseline-v2/config.json", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "climbmix-d12-r12-baseline-v2", - "experiment": { - "experiment_id": "climbmix-d12-r12-baseline-v2", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 8, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "storage": { - "hf_model_repo": "jbduran/think-nanochat-d12", - "path": "experiments/climbmix-d12-r12-baseline-v2" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "climbmix-d12-r12-baseline-v2", - "group": "climbmix-d12", - "tags": [ - "climbmix", - "d12", - "baseline", - "ratio12" - ] - } - } - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 3, - "pos": 55566403, - "epoch": 2, - "pq_idx": 3, - "rg_idx": 55566403 - }, - "loop_state": { - "min_val_bpb": 0.9399637061635419, - "smooth_train_loss": 3.0342439616648695, - "total_training_time": 3982.5033581256866 - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/model_000500.pt b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/model_000500.pt deleted file mode 100644 index eed07a195a2781713ea33809a0096953592d9c42..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:85c521a72ea4400bdbe8b2eb173deae14f7957a9075268bbbd1a0d4690223b60 -size 792761690 diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/model_001000.pt b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/model_001000.pt deleted file mode 100644 index 833db2f1d54bec2a2073647f8ec6c762a090f4b2..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e3c2f15fb478ba9da622f478a05f978c8f156c9c19a192c957051b188a8c83b0 -size 792761690 diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/model_001500.pt b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/model_001500.pt deleted file mode 100644 index 9700d4492365a98eaa2f8ba362e78f511d1ed7f7..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:be2b9308d65dc1687b9f470e9c0fccba96c13c4a5c38039d12ab9d3a1f7b13b3 -size 792761690 diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/optim_000500_rank0.pt b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 023786dbc6305c65125908078cfbd4bbbe8b1519..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8bcf35aa0cb4c98428faf8b56dc0fc6a3452cf542b63afa7824ef6cc31466fcc -size 1246165357 diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/optim_001000_rank0.pt b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 66b53a833f54401c7ca504966d56a48d42c23d76..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6ec90a53b26de44c0a274953ea413bb475101d9c0d0047496a41b8c80ba03da2 -size 1246165357 diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/optim_001500_rank0.pt b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 38301921dce3a0da18ffed43faaa22fe67fd9861..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e15eba406d45bb7091f66c97fa893a4267fc04ef004f5f720fae7786544dc7bc -size 1246165357 diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/config.json b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/config.json deleted file mode 100644 index 778ce5a90e8c4bd0d532bce0302c6edcabedf8b9..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/config.json +++ /dev/null @@ -1,54 +0,0 @@ -{ - "experiment_id": "climbmix-d12-r12-baseline-v2", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 8, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "storage": { - "hf_model_repo": "jbduran/think-nanochat-d12", - "path": "experiments/climbmix-d12-r12-baseline-v2" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "climbmix-d12-r12-baseline-v2", - "group": "climbmix-d12", - "tags": [ - "climbmix", - "d12", - "baseline", - "ratio12" - ] - } -} diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/run.json b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/run.json deleted file mode 100644 index 20e477094c9df384bc7c53839540c7c9db3607a9..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/run.json +++ /dev/null @@ -1,5 +0,0 @@ -{ - "experiment_id": "climbmix-d12-r12-baseline-v2", - "wandb_run_id": "57051863", - "created_at": 1781140862 -} diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/tokenizer/experiment_tokenizer.json b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 1daeeb9f11dbc461b27c26c41d062c1473690bb2..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,19 +0,0 @@ -{ - "experiment_id": "climbmix-d12-r12-baseline-v2", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 8, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1781120899 -} diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/tokenizer/token_bytes.pt b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/tokenizer/token_bytes.pt deleted file mode 100644 index 4225802199c2b8e53978b8feb8b1fe4f19f036fb..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:398732dad177888c6f884c3abb76168a6b54fd7c3f011b2d38371587faca54b7 -size 132649 diff --git a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/tokenizer/tokenizer.pkl b/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/tokenizer/tokenizer.pkl deleted file mode 100644 index b805582aa68ae9ef75e9bac2df1eb87ba287a958..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/experiments/climbmix-d12-r12-baseline-v2/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:387cfc082b0bee45467774fd6f1310a922ad170886a58ccddcb468f275e06a6c -size 412105 diff --git a/archive/pre-lineage-v1/tokenizer/think_dataset_tokenizer.json b/archive/pre-lineage-v1/tokenizer/think_dataset_tokenizer.json deleted file mode 100644 index fea3141f8940ad53370916cd7a3dc8743edbf3ad..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/tokenizer/think_dataset_tokenizer.json +++ /dev/null @@ -1,20 +0,0 @@ -{ - "dataset_repo": "jbduran/think-dataset", - "manifest_source_dataset": "institutional/institutional-books-1.0", - "num_train_shards": 24, - "expected_val_shard": "shard_00472.parquet", - "filters": { - "language": "eng", - "min_english_proportion": 0.9, - "year_max_exclusive": 1930, - "reject_invalid_date_types": true, - "ocr_min_inclusive": 90.0, - "ocr_disagreement_max_inclusive": 10.0, - "min_tokenizability": 95.0, - "min_tokens": 500, - "min_chars": 2000, - "min_pages": 3, - "min_sentences": 20, - "undated_rows_rejected": true - } -} \ No newline at end of file diff --git a/archive/pre-lineage-v1/tokenizer/token_bytes.pt b/archive/pre-lineage-v1/tokenizer/token_bytes.pt deleted file mode 100644 index 01d1ec4aab9e8a7d205c3b3ffbeb8da0e9a62db1..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1 -size 132649 diff --git a/archive/pre-lineage-v1/tokenizer/tokenizer.pkl b/archive/pre-lineage-v1/tokenizer/tokenizer.pkl deleted file mode 100644 index a17bd392980021628053b95d6425fc556aad527a..0000000000000000000000000000000000000000 --- a/archive/pre-lineage-v1/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1 -size 404071 diff --git a/evaluations/vintage-core-v1.0.0/_runner.json b/evaluations/vintage-core-v1.0.0/_runner.json deleted file mode 100644 index 6e765056959c355595420be3ae565eb38146f09b..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/_runner.json +++ /dev/null @@ -1,13 +0,0 @@ -{ - "schema_version": 1, - "evaluator_commit": "82b7e92adf04aac6418b29e6bbca7ddfd479c462", - "models": [ - "modern-d24", - "gpt1900-d34" - ], - "bundles": [ - "original", - "filtered", - "restyled" - ] -} \ No newline at end of file diff --git a/evaluations/vintage-core-v1.0.0/gpt1900-d34/filtered.json b/evaluations/vintage-core-v1.0.0/gpt1900-d34/filtered.json deleted file mode 100644 index 19cd218efe3d4ca4ac91298cf0d0826cbfe9f2cf..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/gpt1900-d34/filtered.json +++ /dev/null @@ -1,51 +0,0 @@ -{ - "bundle": "filtered", - "centered_results": { - "agi_eval_lsat_ar": 0.010869565217391276, - "arc_challenge": 0.010238907849829356, - "arc_easy": 0.24686468646864687, - "bigbench_language_identification": 0.1773153537282394, - "bigbench_operators": 0.12857142857142856, - "bigbench_qa_wikidata": 0.37925957088767354, - "bigbench_repeat_copy_logic": 0.0, - "boolq": -0.46718146718146725, - "commonsense_qa": 0.04995904995904993, - "copa": 0.15999999999999992, - "coqa": 0.23559718969555035, - "hellaswag": 0.1479043230195304, - "hellaswag_zeroshot": 0.14417379855167875, - "jeopardy": 0.047619047619047616, - "lambada_openai": 0.4005014816503305, - "openbook_qa": 0.061333333333333316, - "piqa": 0.19332827899924188, - "squad": 0.1388888888888889, - "winograd": 0.3846153846153846, - "winogrande": 0.09707971586424624 - }, - "core_metric": 0.12734692688690122, - "max_per_task": -1, - "model": "gpt1900-d34", - "results": { - "agi_eval_lsat_ar": 0.20869565217391303, - "arc_challenge": 0.257679180887372, - "arc_easy": 0.43514851485148515, - "bigbench_language_identification": 0.2521796565389696, - "bigbench_operators": 0.12857142857142856, - "bigbench_qa_wikidata": 0.37925957088767354, - "bigbench_repeat_copy_logic": 0.0, - "boolq": 0.45714285714285713, - "commonsense_qa": 0.23996723996723995, - "copa": 0.58, - "coqa": 0.23559718969555035, - "hellaswag": 0.3609282422646478, - "hellaswag_zeroshot": 0.35813034891375906, - "jeopardy": 0.047619047619047616, - "lambada_openai": 0.4005014816503305, - "openbook_qa": 0.296, - "piqa": 0.5966641394996209, - "squad": 0.1388888888888889, - "winograd": 0.6923076923076923, - "winogrande": 0.5485398579321231 - }, - "runtime_seconds": 2171.2343723773956 -} diff --git a/evaluations/vintage-core-v1.0.0/gpt1900-d34/original.json b/evaluations/vintage-core-v1.0.0/gpt1900-d34/original.json deleted file mode 100644 index 654331e9ecbbefd3481bb26069b802fc0d57acd7..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/gpt1900-d34/original.json +++ /dev/null @@ -1,55 +0,0 @@ -{ - "bundle": "original", - "centered_results": { - "agi_eval_lsat_ar": 0.0815217391304348, - "arc_challenge": 0.0011376564277588337, - "arc_easy": 0.2171717171717172, - "bigbench_cs_algorithms": 0.3696969696969697, - "bigbench_dyck_languages": 0.133, - "bigbench_language_identification": 0.18261826182618263, - "bigbench_operators": 0.12857142857142856, - "bigbench_qa_wikidata": 0.28719059101422173, - "bigbench_repeat_copy_logic": 0.0, - "boolq": -0.39948495090938346, - "commonsense_qa": 0.053030303030303025, - "copa": 0.17999999999999994, - "coqa": 0.21821370412125768, - "hellaswag": 0.12394609307574855, - "hellaswag_zeroshot": 0.12381331740025225, - "jeopardy": 0.03306565895134624, - "lambada_openai": 0.3968562002716864, - "openbook_qa": 0.040000000000000036, - "piqa": 0.16213275299238306, - "squad": 0.1269631031220435, - "winograd": 0.3919413919413919, - "winogrande": 0.07971586424625099 - }, - "core_metric": 0.13323190009463606, - "max_per_task": -1, - "model": "gpt1900-d34", - "results": { - "agi_eval_lsat_ar": 0.26521739130434785, - "arc_challenge": 0.2508532423208191, - "arc_easy": 0.4128787878787879, - "bigbench_cs_algorithms": 0.3696969696969697, - "bigbench_dyck_languages": 0.133, - "bigbench_language_identification": 0.257, - "bigbench_operators": 0.12857142857142856, - "bigbench_qa_wikidata": 0.28719059101422173, - "bigbench_repeat_copy_logic": 0.0, - "boolq": 0.46819571865443427, - "commonsense_qa": 0.24242424242424243, - "copa": 0.59, - "coqa": 0.21821370412125768, - "hellaswag": 0.3429595698068114, - "hellaswag_zeroshot": 0.3428599880501892, - "jeopardy": 0.03306565895134624, - "lambada_openai": 0.3968562002716864, - "openbook_qa": 0.28, - "piqa": 0.5810663764961915, - "squad": 0.1269631031220435, - "winograd": 0.6959706959706959, - "winogrande": 0.5398579321231255 - }, - "runtime_seconds": 3620.568478822708 -} diff --git a/evaluations/vintage-core-v1.0.0/gpt1900-d34/restyled.json b/evaluations/vintage-core-v1.0.0/gpt1900-d34/restyled.json deleted file mode 100644 index 494a671f6e063a6a030cf78bbf89c77a205be8d3..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/gpt1900-d34/restyled.json +++ /dev/null @@ -1,51 +0,0 @@ -{ - "bundle": "restyled", - "centered_results": { - "agi_eval_lsat_ar": 0.010869565217391276, - "arc_challenge": 0.018202502844141044, - "arc_easy": 0.2495049504950495, - "bigbench_language_identification": 0.1773153537282394, - "bigbench_operators": 0.12857142857142856, - "bigbench_qa_wikidata": 0.37925957088767354, - "bigbench_repeat_copy_logic": 0.03125, - "boolq": -0.4804952735987219, - "commonsense_qa": 0.046887796887796866, - "copa": 0.24, - "coqa": 0.23278688524590163, - "hellaswag": 0.15141540487162608, - "hellaswag_zeroshot": 0.1500987491770902, - "jeopardy": 0.05128205128205128, - "lambada_openai": 0.4009573740597219, - "openbook_qa": 0.072, - "piqa": 0.200909780136467, - "squad": 0.1370214752567694, - "winograd": 0.4578754578754578, - "winogrande": 0.1176006314127862 - }, - "core_metric": 0.13866568521754347, - "max_per_task": -1, - "model": "gpt1900-d34", - "results": { - "agi_eval_lsat_ar": 0.20869565217391303, - "arc_challenge": 0.2636518771331058, - "arc_easy": 0.43712871287128713, - "bigbench_language_identification": 0.2521796565389696, - "bigbench_operators": 0.12857142857142856, - "bigbench_qa_wikidata": 0.37925957088767354, - "bigbench_repeat_copy_logic": 0.03125, - "boolq": 0.4522167487684729, - "commonsense_qa": 0.2375102375102375, - "copa": 0.62, - "coqa": 0.23278688524590163, - "hellaswag": 0.36356155365371956, - "hellaswag_zeroshot": 0.36257406188281766, - "jeopardy": 0.05128205128205128, - "lambada_openai": 0.4009573740597219, - "openbook_qa": 0.304, - "piqa": 0.6004548900682335, - "squad": 0.1370214752567694, - "winograd": 0.7289377289377289, - "winogrande": 0.5588003157063931 - }, - "runtime_seconds": 2194.441705226898 -} diff --git a/evaluations/vintage-core-v1.0.0/gpt1900-d34/summary.csv b/evaluations/vintage-core-v1.0.0/gpt1900-d34/summary.csv deleted file mode 100644 index 2ef312e427e1d1ea2a57bacbd58693a7c2ab0111..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/gpt1900-d34/summary.csv +++ /dev/null @@ -1,4 +0,0 @@ -model,bundle,native_core,common_20_core,runtime_seconds -gpt1900-d34,original,0.13323190009463606,0.12142024161925118,3620.568478822708 -gpt1900-d34,filtered,0.12734692688690122,0.1273469268869012,2171.2343723773956 -gpt1900-d34,restyled,0.13866568521754347,0.1386656852175435,2194.441705226898 diff --git a/evaluations/vintage-core-v1.0.0/gpt1900-d34/task_accuracy.csv b/evaluations/vintage-core-v1.0.0/gpt1900-d34/task_accuracy.csv deleted file mode 100644 index 20159d6d825f362aeb20bfa09f85b8a7c0bd7c17..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/gpt1900-d34/task_accuracy.csv +++ /dev/null @@ -1,23 +0,0 @@ -task,original,filtered,restyled -agi_eval_lsat_ar,0.26521739130434785,0.20869565217391303,0.20869565217391303 -arc_challenge,0.2508532423208191,0.257679180887372,0.2636518771331058 -arc_easy,0.4128787878787879,0.43514851485148515,0.43712871287128713 -bigbench_cs_algorithms,0.3696969696969697,, -bigbench_dyck_languages,0.133,, -bigbench_language_identification,0.257,0.2521796565389696,0.2521796565389696 -bigbench_operators,0.12857142857142856,0.12857142857142856,0.12857142857142856 -bigbench_qa_wikidata,0.28719059101422173,0.37925957088767354,0.37925957088767354 -bigbench_repeat_copy_logic,0.0,0.0,0.03125 -boolq,0.46819571865443427,0.45714285714285713,0.4522167487684729 -commonsense_qa,0.24242424242424243,0.23996723996723995,0.2375102375102375 -copa,0.59,0.58,0.62 -coqa,0.21821370412125768,0.23559718969555035,0.23278688524590163 -hellaswag,0.3429595698068114,0.3609282422646478,0.36356155365371956 -hellaswag_zeroshot,0.3428599880501892,0.35813034891375906,0.36257406188281766 -jeopardy,0.03306565895134624,0.047619047619047616,0.05128205128205128 -lambada_openai,0.3968562002716864,0.4005014816503305,0.4009573740597219 -openbook_qa,0.28,0.296,0.304 -piqa,0.5810663764961915,0.5966641394996209,0.6004548900682335 -squad,0.1269631031220435,0.1388888888888889,0.1370214752567694 -winograd,0.6959706959706959,0.6923076923076923,0.7289377289377289 -winogrande,0.5398579321231255,0.5485398579321231,0.5588003157063931 diff --git a/evaluations/vintage-core-v1.0.0/gpt1900-d34/task_deltas.csv b/evaluations/vintage-core-v1.0.0/gpt1900-d34/task_deltas.csv deleted file mode 100644 index 765d2b33ade8143fe3ebc8a3cd361aea64557f81..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/gpt1900-d34/task_deltas.csv +++ /dev/null @@ -1,21 +0,0 @@ -task,filtered_minus_original,restyled_minus_filtered -agi_eval_lsat_ar,-0.05652173913043482,0.0 -arc_challenge,0.0068259385665528916,0.005972696245733766 -arc_easy,0.022269726972697246,0.001980198019801982 -bigbench_language_identification,-0.004820343461030385,0.0 -bigbench_operators,0.0,0.0 -bigbench_qa_wikidata,0.0920689798734518,0.0 -bigbench_repeat_copy_logic,0.0,0.03125 -boolq,-0.011052861511577139,-0.0049261083743842304 -commonsense_qa,-0.0024570024570024773,-0.0024570024570024496 -copa,-0.010000000000000009,0.040000000000000036 -coqa,0.017383485574292673,-0.0028103044496487206 -hellaswag,0.0179686724578364,0.0026333113890717463 -hellaswag_zeroshot,0.015270360863569865,0.0044437129690586 -jeopardy,0.014553388667701374,0.003663003663003664 -lambada_openai,0.0036452813786441163,0.00045589240939142295 -openbook_qa,0.01599999999999996,0.008000000000000007 -piqa,0.01559776300342941,0.0037907505686125553 -squad,0.011925785766845387,-0.0018674136321195078 -winograd,-0.00366300366300365,0.03663003663003661 -winogrande,0.008681925808997626,0.010260457774269982 diff --git a/evaluations/vintage-core-v1.0.0/gpt1900-d34/worker_input.json b/evaluations/vintage-core-v1.0.0/gpt1900-d34/worker_input.json deleted file mode 100644 index 222dae3382fd03424aaf346ad207c837144f0b04..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/gpt1900-d34/worker_input.json +++ /dev/null @@ -1,31 +0,0 @@ -{ - "bundles": { - "filtered": "/workspace/nanochat/reference-evals/cache/huggingface/datasets--jbduran--vintage-core/snapshots/7807a9d9f2f3cf1425289a2810b4fa8c813f3c2c/filtered", - "original": "/workspace/nanochat/reference-evals/cache/bundles/original/extracted/eval_bundle", - "restyled": "/workspace/nanochat/reference-evals/cache/huggingface/datasets--jbduran--vintage-core/snapshots/7807a9d9f2f3cf1425289a2810b4fa8c813f3c2c/restyled" - }, - "max_per_task": -1, - "model": { - "allow_patterns": [ - "model_010507.pt", - "meta_010507.json", - "tokenizer/tokenizer.pkl", - "nanochat/**" - ], - "artifact_repo": "mhla/gpt1900-d34-22btok", - "artifact_revision": "d6330f9f0a17ce13da36fb951d7987bb03e6fbd0", - "checkpoint": "model_010507.pt", - "display_name": "GPT-1900 d34 22B-token", - "metadata": "meta_010507.json", - "recommended_gpu": "A100-class GPU required", - "runtime": { - "path": ".", - "type": "bundled" - }, - "tokenizer_dir": "tokenizer" - }, - "model_id": "gpt1900-d34", - "output_dir": "/workspace/nanochat/reference-evals/vintage-core-v1.0.0/gpt1900-d34", - "runtime": "/workspace/nanochat/reference-evals/cache/huggingface/models--mhla--gpt1900-d34-22btok/snapshots/d6330f9f0a17ce13da36fb951d7987bb03e6fbd0", - "snapshot": "/workspace/nanochat/reference-evals/cache/huggingface/models--mhla--gpt1900-d34-22btok/snapshots/d6330f9f0a17ce13da36fb951d7987bb03e6fbd0" -} diff --git a/evaluations/vintage-core-v1.0.0/modern-d24/filtered.json b/evaluations/vintage-core-v1.0.0/modern-d24/filtered.json deleted file mode 100644 index 87fa1f7126fabe6baa0b1838dd5e3c702821326e..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/modern-d24/filtered.json +++ /dev/null @@ -1,51 +0,0 @@ -{ - "bundle": "filtered", - "centered_results": { - "agi_eval_lsat_ar": 0.10326086956521735, - "arc_challenge": 0.18430034129692832, - "arc_easy": 0.570957095709571, - "bigbench_language_identification": 0.1765887288860986, - "bigbench_operators": 0.17142857142857143, - "bigbench_qa_wikidata": 0.5623685317627262, - "bigbench_repeat_copy_logic": 0.0, - "boolq": -0.176940487285315, - "commonsense_qa": 0.1554054054054054, - "copa": 0.3400000000000001, - "coqa": 0.23770491803278687, - "hellaswag": 0.3570331358349792, - "hellaswag_zeroshot": 0.35418038183015144, - "jeopardy": 0.19047619047619047, - "lambada_openai": 0.4253476179621609, - "openbook_qa": 0.2293333333333333, - "piqa": 0.42228961334344195, - "squad": 0.34780578898225956, - "winograd": 0.4358974358974359, - "winogrande": 0.1428571428571428 - }, - "core_metric": 0.26151473076595433, - "max_per_task": -1, - "model": "modern-d24", - "results": { - "agi_eval_lsat_ar": 0.2826086956521739, - "arc_challenge": 0.38822525597269625, - "arc_easy": 0.6782178217821783, - "bigbench_language_identification": 0.25151915455746365, - "bigbench_operators": 0.17142857142857143, - "bigbench_qa_wikidata": 0.5623685317627262, - "bigbench_repeat_copy_logic": 0.0, - "boolq": 0.5645320197044335, - "commonsense_qa": 0.32432432432432434, - "copa": 0.67, - "coqa": 0.23770491803278687, - "hellaswag": 0.5177748518762344, - "hellaswag_zeroshot": 0.5156352863726136, - "jeopardy": 0.19047619047619047, - "lambada_openai": 0.4253476179621609, - "openbook_qa": 0.422, - "piqa": 0.711144806671721, - "squad": 0.34780578898225956, - "winograd": 0.717948717948718, - "winogrande": 0.5714285714285714 - }, - "runtime_seconds": 1260.3533561229706 -} diff --git a/evaluations/vintage-core-v1.0.0/modern-d24/original.json b/evaluations/vintage-core-v1.0.0/modern-d24/original.json deleted file mode 100644 index b887b984d04a56f44383c54ba2d614793fcba1b2..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/modern-d24/original.json +++ /dev/null @@ -1,55 +0,0 @@ -{ - "bundle": "original", - "centered_results": { - "agi_eval_lsat_ar": 0.13043478260869565, - "arc_challenge": 0.1808873720136519, - "arc_easy": 0.5695847362514029, - "bigbench_cs_algorithms": 0.4166666666666667, - "bigbench_dyck_languages": 0.123, - "bigbench_language_identification": 0.17447744774477447, - "bigbench_operators": 0.17142857142857143, - "bigbench_qa_wikidata": 0.5106047930712071, - "bigbench_repeat_copy_logic": 0.0, - "boolq": -0.09206502494769035, - "commonsense_qa": 0.14721539721539723, - "copa": 0.3400000000000001, - "coqa": 0.254666165601904, - "hellaswag": 0.3572329549226582, - "hellaswag_zeroshot": 0.35617074951868827, - "jeopardy": 0.19272555503070382, - "lambada_openai": 0.42712982728507665, - "openbook_qa": 0.20000000000000004, - "piqa": 0.427638737758433, - "squad": 0.3544938505203406, - "winograd": 0.4212454212454213, - "winogrande": 0.1333859510655091 - }, - "core_metric": 0.2634965434091551, - "max_per_task": -1, - "model": "modern-d24", - "results": { - "agi_eval_lsat_ar": 0.30434782608695654, - "arc_challenge": 0.3856655290102389, - "arc_easy": 0.6771885521885522, - "bigbench_cs_algorithms": 0.4166666666666667, - "bigbench_dyck_languages": 0.123, - "bigbench_language_identification": 0.2496, - "bigbench_operators": 0.17142857142857143, - "bigbench_qa_wikidata": 0.5106047930712071, - "bigbench_repeat_copy_logic": 0.0, - "boolq": 0.5850152905198777, - "commonsense_qa": 0.3177723177723178, - "copa": 0.67, - "coqa": 0.254666165601904, - "hellaswag": 0.5179247161919937, - "hellaswag_zeroshot": 0.5171280621390162, - "jeopardy": 0.19272555503070382, - "lambada_openai": 0.42712982728507665, - "openbook_qa": 0.4, - "piqa": 0.7138193688792165, - "squad": 0.3544938505203406, - "winograd": 0.7106227106227107, - "winogrande": 0.5666929755327546 - }, - "runtime_seconds": 2120.226320743561 -} diff --git a/evaluations/vintage-core-v1.0.0/modern-d24/restyled.json b/evaluations/vintage-core-v1.0.0/modern-d24/restyled.json deleted file mode 100644 index 68d8b3d9a0cc8e092bb203378fa6bc2628dcd452..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/modern-d24/restyled.json +++ /dev/null @@ -1,51 +0,0 @@ -{ - "bundle": "restyled", - "centered_results": { - "agi_eval_lsat_ar": 0.10326086956521735, - "arc_challenge": 0.17974971558589306, - "arc_easy": 0.537953795379538, - "bigbench_language_identification": 0.1765887288860986, - "bigbench_operators": 0.17142857142857143, - "bigbench_qa_wikidata": 0.5623685317627262, - "bigbench_repeat_copy_logic": 0.0, - "boolq": -0.1609639195846092, - "commonsense_qa": 0.15233415233415232, - "copa": 0.32000000000000006, - "coqa": 0.23442622950819672, - "hellaswag": 0.33596664472240506, - "hellaswag_zeroshot": 0.34364713627386445, - "jeopardy": 0.18986568986568986, - "lambada_openai": 0.4264873489856394, - "openbook_qa": 0.2213333333333333, - "piqa": 0.4010614101592116, - "squad": 0.34360410830999066, - "winograd": 0.4065934065934067, - "winogrande": 0.1176006314127862 - }, - "core_metric": 0.25316531922610563, - "max_per_task": -1, - "model": "modern-d24", - "results": { - "agi_eval_lsat_ar": 0.2826086956521739, - "arc_challenge": 0.3848122866894198, - "arc_easy": 0.6534653465346535, - "bigbench_language_identification": 0.25151915455746365, - "bigbench_operators": 0.17142857142857143, - "bigbench_qa_wikidata": 0.5623685317627262, - "bigbench_repeat_copy_logic": 0.0, - "boolq": 0.5704433497536946, - "commonsense_qa": 0.32186732186732187, - "copa": 0.66, - "coqa": 0.23442622950819672, - "hellaswag": 0.5019749835418038, - "hellaswag_zeroshot": 0.5077353522053983, - "jeopardy": 0.18986568986568986, - "lambada_openai": 0.4264873489856394, - "openbook_qa": 0.416, - "piqa": 0.7005307050796058, - "squad": 0.34360410830999066, - "winograd": 0.7032967032967034, - "winogrande": 0.5588003157063931 - }, - "runtime_seconds": 1268.2236154079437 -} diff --git a/evaluations/vintage-core-v1.0.0/modern-d24/summary.csv b/evaluations/vintage-core-v1.0.0/modern-d24/summary.csv deleted file mode 100644 index 07000fddeb4792ceb23e0a73e9d22d6232a60363..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/modern-d24/summary.csv +++ /dev/null @@ -1,4 +0,0 @@ -model,bundle,native_core,common_20_core,runtime_seconds -modern-d24,original,0.2634965434091551,0.2628628644167373,2120.226320743561 -modern-d24,filtered,0.26151473076595433,0.26151473076595433,1260.3533561229706 -modern-d24,restyled,0.25316531922610563,0.25316531922610563,1268.2236154079437 diff --git a/evaluations/vintage-core-v1.0.0/modern-d24/task_accuracy.csv b/evaluations/vintage-core-v1.0.0/modern-d24/task_accuracy.csv deleted file mode 100644 index cbc7454a12d4272a47a8f2c0c41724fe0860bad5..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/modern-d24/task_accuracy.csv +++ /dev/null @@ -1,23 +0,0 @@ -task,original,filtered,restyled -agi_eval_lsat_ar,0.30434782608695654,0.2826086956521739,0.2826086956521739 -arc_challenge,0.3856655290102389,0.38822525597269625,0.3848122866894198 -arc_easy,0.6771885521885522,0.6782178217821783,0.6534653465346535 -bigbench_cs_algorithms,0.4166666666666667,, -bigbench_dyck_languages,0.123,, -bigbench_language_identification,0.2496,0.25151915455746365,0.25151915455746365 -bigbench_operators,0.17142857142857143,0.17142857142857143,0.17142857142857143 -bigbench_qa_wikidata,0.5106047930712071,0.5623685317627262,0.5623685317627262 -bigbench_repeat_copy_logic,0.0,0.0,0.0 -boolq,0.5850152905198777,0.5645320197044335,0.5704433497536946 -commonsense_qa,0.3177723177723178,0.32432432432432434,0.32186732186732187 -copa,0.67,0.67,0.66 -coqa,0.254666165601904,0.23770491803278687,0.23442622950819672 -hellaswag,0.5179247161919937,0.5177748518762344,0.5019749835418038 -hellaswag_zeroshot,0.5171280621390162,0.5156352863726136,0.5077353522053983 -jeopardy,0.19272555503070382,0.19047619047619047,0.18986568986568986 -lambada_openai,0.42712982728507665,0.4253476179621609,0.4264873489856394 -openbook_qa,0.4,0.422,0.416 -piqa,0.7138193688792165,0.711144806671721,0.7005307050796058 -squad,0.3544938505203406,0.34780578898225956,0.34360410830999066 -winograd,0.7106227106227107,0.717948717948718,0.7032967032967034 -winogrande,0.5666929755327546,0.5714285714285714,0.5588003157063931 diff --git a/evaluations/vintage-core-v1.0.0/modern-d24/task_deltas.csv b/evaluations/vintage-core-v1.0.0/modern-d24/task_deltas.csv deleted file mode 100644 index 09f859d234ccf559dd09b09c11398bc09029f2d9..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/modern-d24/task_deltas.csv +++ /dev/null @@ -1,21 +0,0 @@ -task,filtered_minus_original,restyled_minus_filtered -agi_eval_lsat_ar,-0.02173913043478265,0.0 -arc_challenge,0.0025597269624573205,-0.0034129692832764458 -arc_easy,0.0010292695936260365,-0.024752475247524774 -bigbench_language_identification,0.0019191545574636648,0.0 -bigbench_operators,0.0,0.0 -bigbench_qa_wikidata,0.05176373869151907,0.0 -bigbench_repeat_copy_logic,0.0,0.0 -boolq,-0.02048327081544421,0.005911330049261143 -commonsense_qa,0.006552006552006551,-0.0024570024570024773 -copa,0.0,-0.010000000000000009 -coqa,-0.016961247569117155,-0.003278688524590151 -hellaswag,-0.00014986431575925163,-0.01579986833443059 -hellaswag_zeroshot,-0.0014927757664026098,-0.007899934167215239 -jeopardy,-0.0022493645545133556,-0.0006105006105006083 -lambada_openai,-0.0017822093229157288,0.001139731023478474 -openbook_qa,0.021999999999999964,-0.006000000000000005 -piqa,-0.002674562207495512,-0.010614101592115177 -squad,-0.006688061538081047,-0.004201680672268893 -winograd,0.0073260073260073,-0.0146520146520146 -winogrande,0.0047355958958168465,-0.012628255722178294 diff --git a/evaluations/vintage-core-v1.0.0/modern-d24/worker_input.json b/evaluations/vintage-core-v1.0.0/modern-d24/worker_input.json deleted file mode 100644 index 67ae9a24c4f13bae045006c51a486bfb8339a1cd..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/modern-d24/worker_input.json +++ /dev/null @@ -1,32 +0,0 @@ -{ - "bundles": { - "filtered": "/workspace/nanochat/reference-evals/cache/huggingface/datasets--jbduran--vintage-core/snapshots/7807a9d9f2f3cf1425289a2810b4fa8c813f3c2c/filtered", - "original": "/workspace/nanochat/reference-evals/cache/bundles/original/extracted/eval_bundle", - "restyled": "/workspace/nanochat/reference-evals/cache/huggingface/datasets--jbduran--vintage-core/snapshots/7807a9d9f2f3cf1425289a2810b4fa8c813f3c2c/restyled" - }, - "max_per_task": -1, - "model": { - "allow_patterns": [ - "model_016704.pt", - "meta_016704.json", - "tokenizer/tokenizer.pkl" - ], - "artifact_repo": "ChrisMcCormick/nanochat-d24-2026-02-02", - "artifact_revision": "2ccf42323ff3bedcc986191d688e5827f33c237c", - "checkpoint": "model_016704.pt", - "display_name": "nanochat modern d24", - "metadata": "meta_016704.json", - "published_original_core": 0.2633, - "recommended_gpu": "A100 recommended", - "runtime": { - "revision": "7ac837cff8efc0e85502e2b3a934a35e2d937b8d", - "type": "git", - "url": "https://github.com/chrisjmccormick/nanochat.git" - }, - "tokenizer_dir": "tokenizer" - }, - "model_id": "modern-d24", - "output_dir": "/workspace/nanochat/reference-evals/vintage-core-v1.0.0/modern-d24", - "runtime": "/workspace/nanochat/reference-evals/cache/runtimes/ChrisMcCormick--nanochat-d24-2026-02-02-7ac837cff8efc0e85502e2b3a934a35e2d937b8d", - "snapshot": "/workspace/nanochat/reference-evals/cache/huggingface/models--ChrisMcCormick--nanochat-d24-2026-02-02/snapshots/2ccf42323ff3bedcc986191d688e5827f33c237c" -} diff --git a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/filtered.json b/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/filtered.json deleted file mode 100644 index 4e0c63df756111954effb8dedc7d93def2cb6953..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/filtered.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "bundle": "filtered", - "centered_results": { - "agi_eval_lsat_ar": -0.0054347826086956555, - "arc_challenge": 0.22525597269624575, - "arc_easy": 0.5425742574257426, - "bigbench_language_identification": 0.17542612913867342, - "bigbench_operators": 0.24285714285714285, - "bigbench_qa_wikidata": 0.6157972233908288, - "bigbench_repeat_copy_logic": 0.0625, - "boolq": 0.26241512448408993, - "commonsense_qa": 0.06531531531531527, - "copa": 0.5, - "coqa": 0.42857142857142855, - "hellaswag": 0.41562431424182583, - "hellaswag_zeroshot": 0.3844634628044766, - "jeopardy": 0.398046398046398, - "lambada_openai": 0.610895828584454, - "openbook_qa": 0.15733333333333333, - "piqa": 0.4268385140257771, - "squad": 0.542016806722689, - "winograd": 0.6263736263736264, - "winogrande": 0.3070244672454616 - }, - "core_metric": 0.3491947281324407, - "evaluator_commit": "82b7e92adf04aac6418b29e6bbca7ddfd479c462", - "max_per_task": -1, - "model": "talkie-1930-13b-base", - "model_repo": "talkie-lm/talkie-1930-13b-base", - "model_revision": "b7c97680791f7fca4262c3c80b36ff7d666faab0", - "prepend_endoftext": true, - "results": { - "agi_eval_lsat_ar": 0.1956521739130435, - "arc_challenge": 0.4189419795221843, - "arc_easy": 0.656930693069307, - "bigbench_language_identification": 0.25046235138705414, - "bigbench_operators": 0.24285714285714285, - "bigbench_qa_wikidata": 0.6157972233908288, - "bigbench_repeat_copy_logic": 0.0625, - "boolq": 0.7270935960591133, - "commonsense_qa": 0.25225225225225223, - "copa": 0.75, - "coqa": 0.42857142857142855, - "hellaswag": 0.5617182356813694, - "hellaswag_zeroshot": 0.5383475971033574, - "jeopardy": 0.398046398046398, - "lambada_openai": 0.610895828584454, - "openbook_qa": 0.368, - "piqa": 0.7134192570128886, - "squad": 0.542016806722689, - "winograd": 0.8131868131868132, - "winogrande": 0.6535122336227308 - }, - "runner_sha256": "2e3408617481a587c45fa130e1fce34009c135253a137d15320b05b3e513acde", - "runtime_revision": "35317ba3a84861a84c84065bd73faf88ad19329c", - "runtime_seconds": 5564.70161485672, - "scoring": "native Talkie BF16; continuation mean loss / exact-token LM" -} diff --git a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/original.json b/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/original.json deleted file mode 100644 index 2a5cffb151237d909af26edbaa8b668b765ffcbc..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/original.json +++ /dev/null @@ -1,62 +0,0 @@ -{ - "bundle": "original", - "centered_results": { - "agi_eval_lsat_ar": 0.0652173913043478, - "arc_challenge": 0.18202502844141066, - "arc_easy": 0.48260381593714924, - "bigbench_cs_algorithms": 0.4401515151515151, - "bigbench_dyck_languages": 0.126, - "bigbench_language_identification": 0.17898789878987897, - "bigbench_operators": 0.24285714285714285, - "bigbench_qa_wikidata": 0.48132473795580927, - "bigbench_repeat_copy_logic": 0.0625, - "boolq": 0.2813455657492356, - "commonsense_qa": 0.0745290745290745, - "copa": 0.45999999999999996, - "coqa": 0.43079042966303394, - "hellaswag": 0.388036911637788, - "hellaswag_zeroshot": 0.35603797384319186, - "jeopardy": 0.3334907888521493, - "lambada_openai": 0.6070250339607995, - "openbook_qa": 0.12000000000000004, - "piqa": 0.36561479869423286, - "squad": 0.55279091769158, - "winograd": 0.641025641025641, - "winogrande": 0.2801894238358327 - }, - "core_metric": 0.32511564045090063, - "evaluator_commit": "82b7e92adf04aac6418b29e6bbca7ddfd479c462", - "max_per_task": -1, - "model": "talkie-1930-13b-base", - "model_repo": "talkie-lm/talkie-1930-13b-base", - "model_revision": "b7c97680791f7fca4262c3c80b36ff7d666faab0", - "prepend_endoftext": true, - "results": { - "agi_eval_lsat_ar": 0.25217391304347825, - "arc_challenge": 0.386518771331058, - "arc_easy": 0.6119528619528619, - "bigbench_cs_algorithms": 0.4401515151515151, - "bigbench_dyck_languages": 0.126, - "bigbench_language_identification": 0.2537, - "bigbench_operators": 0.24285714285714285, - "bigbench_qa_wikidata": 0.48132473795580927, - "bigbench_repeat_copy_logic": 0.0625, - "boolq": 0.7269113149847095, - "commonsense_qa": 0.2596232596232596, - "copa": 0.73, - "coqa": 0.43079042966303394, - "hellaswag": 0.541027683728341, - "hellaswag_zeroshot": 0.5170284803823939, - "jeopardy": 0.3334907888521493, - "lambada_openai": 0.6070250339607995, - "openbook_qa": 0.34, - "piqa": 0.6828073993471164, - "squad": 0.55279091769158, - "winograd": 0.8205128205128205, - "winogrande": 0.6400947119179163 - }, - "runner_sha256": "61a1f24ba28e09bc347df425da0c5bbaa1512d5aeefa8c8917fefcd8931de42e", - "runtime_revision": "35317ba3a84861a84c84065bd73faf88ad19329c", - "runtime_seconds": 9192.917335748672, - "scoring": "native Talkie BF16; continuation mean loss / exact-token LM" -} diff --git a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/progress/filtered.json b/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/progress/filtered.json deleted file mode 100644 index 8ea5e43eab4f3dc5d12c4bec875c93a5dbc5b6bc..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/progress/filtered.json +++ /dev/null @@ -1,77 +0,0 @@ -{ - "bundle": "filtered", - "centered_results": { - "agi_eval_lsat_ar": -0.0054347826086956555, - "arc_challenge": 0.22525597269624575, - "arc_easy": 0.5425742574257426, - "bigbench_language_identification": 0.17542612913867342, - "bigbench_operators": 0.24285714285714285, - "bigbench_qa_wikidata": 0.6157972233908288, - "bigbench_repeat_copy_logic": 0.0625, - "boolq": 0.26241512448408993, - "commonsense_qa": 0.06531531531531527, - "copa": 0.5, - "coqa": 0.42857142857142855, - "hellaswag": 0.41562431424182583, - "hellaswag_zeroshot": 0.3844634628044766, - "jeopardy": 0.398046398046398, - "lambada_openai": 0.610895828584454, - "openbook_qa": 0.15733333333333333, - "piqa": 0.4268385140257771, - "squad": 0.542016806722689, - "winograd": 0.6263736263736264, - "winogrande": 0.3070244672454616 - }, - "evaluator_commit": "82b7e92adf04aac6418b29e6bbca7ddfd479c462", - "max_per_task": -1, - "model": "talkie-1930-13b-base", - "model_revision": "b7c97680791f7fca4262c3c80b36ff7d666faab0", - "results": { - "agi_eval_lsat_ar": 0.1956521739130435, - "arc_challenge": 0.4189419795221843, - "arc_easy": 0.656930693069307, - "bigbench_language_identification": 0.25046235138705414, - "bigbench_operators": 0.24285714285714285, - "bigbench_qa_wikidata": 0.6157972233908288, - "bigbench_repeat_copy_logic": 0.0625, - "boolq": 0.7270935960591133, - "commonsense_qa": 0.25225225225225223, - "copa": 0.75, - "coqa": 0.42857142857142855, - "hellaswag": 0.5617182356813694, - "hellaswag_zeroshot": 0.5383475971033574, - "jeopardy": 0.398046398046398, - "lambada_openai": 0.610895828584454, - "openbook_qa": 0.368, - "piqa": 0.7134192570128886, - "squad": 0.542016806722689, - "winograd": 0.8131868131868132, - "winogrande": 0.6535122336227308 - }, - "runner_sha256": "2e3408617481a587c45fa130e1fce34009c135253a137d15320b05b3e513acde", - "runtime_revision": "35317ba3a84861a84c84065bd73faf88ad19329c", - "schema_version": 1, - "task_runtime_seconds": { - "agi_eval_lsat_ar": 42.33336901664734, - "arc_challenge": 113.12644743919373, - "arc_easy": 159.9239821434021, - "bigbench_language_identification": 2433.2872059345245, - "bigbench_operators": 6.471834421157837, - "bigbench_qa_wikidata": 292.06081438064575, - "bigbench_repeat_copy_logic": 1.057509422302246, - "boolq": 163.38928627967834, - "commonsense_qa": 133.50174689292908, - "copa": 3.0809524059295654, - "coqa": 147.90537881851196, - "hellaswag": 1083.9819493293762, - "hellaswag_zeroshot": 188.30272459983826, - "jeopardy": 50.05873084068298, - "lambada_openai": 133.78516936302185, - "openbook_qa": 15.524103164672852, - "piqa": 60.12010145187378, - "squad": 490.09943890571594, - "winograd": 8.200031757354736, - "winogrande": 38.490838289260864 - }, - "updated_at_unix": 1786737495.9668913 -} diff --git a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/progress/original.json b/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/progress/original.json deleted file mode 100644 index 699907a63643396eef182784855c58975b102866..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/progress/original.json +++ /dev/null @@ -1,83 +0,0 @@ -{ - "bundle": "original", - "centered_results": { - "agi_eval_lsat_ar": 0.0652173913043478, - "arc_challenge": 0.18202502844141066, - "arc_easy": 0.48260381593714924, - "bigbench_cs_algorithms": 0.4401515151515151, - "bigbench_dyck_languages": 0.126, - "bigbench_language_identification": 0.17898789878987897, - "bigbench_operators": 0.24285714285714285, - "bigbench_qa_wikidata": 0.48132473795580927, - "bigbench_repeat_copy_logic": 0.0625, - "boolq": 0.2813455657492356, - "commonsense_qa": 0.0745290745290745, - "copa": 0.45999999999999996, - "coqa": 0.43079042966303394, - "hellaswag": 0.388036911637788, - "hellaswag_zeroshot": 0.35603797384319186, - "jeopardy": 0.3334907888521493, - "lambada_openai": 0.6070250339607995, - "openbook_qa": 0.12000000000000004, - "piqa": 0.36561479869423286, - "squad": 0.55279091769158, - "winograd": 0.641025641025641, - "winogrande": 0.2801894238358327 - }, - "evaluator_commit": "82b7e92adf04aac6418b29e6bbca7ddfd479c462", - "max_per_task": -1, - "model": "talkie-1930-13b-base", - "model_revision": "b7c97680791f7fca4262c3c80b36ff7d666faab0", - "results": { - "agi_eval_lsat_ar": 0.25217391304347825, - "arc_challenge": 0.386518771331058, - "arc_easy": 0.6119528619528619, - "bigbench_cs_algorithms": 0.4401515151515151, - "bigbench_dyck_languages": 0.126, - "bigbench_language_identification": 0.2537, - "bigbench_operators": 0.24285714285714285, - "bigbench_qa_wikidata": 0.48132473795580927, - "bigbench_repeat_copy_logic": 0.0625, - "boolq": 0.7269113149847095, - "commonsense_qa": 0.2596232596232596, - "copa": 0.73, - "coqa": 0.43079042966303394, - "hellaswag": 0.541027683728341, - "hellaswag_zeroshot": 0.5170284803823939, - "jeopardy": 0.3334907888521493, - "lambada_openai": 0.6070250339607995, - "openbook_qa": 0.34, - "piqa": 0.6828073993471164, - "squad": 0.55279091769158, - "winograd": 0.8205128205128205, - "winogrande": 0.6400947119179163 - }, - "runner_sha256": "61a1f24ba28e09bc347df425da0c5bbaa1512d5aeefa8c8917fefcd8931de42e", - "runtime_revision": "35317ba3a84861a84c84065bd73faf88ad19329c", - "schema_version": 1, - "task_runtime_seconds": { - "agi_eval_lsat_ar": 42.50863289833069, - "arc_challenge": 108.33659100532532, - "arc_easy": 192.92915868759155, - "bigbench_cs_algorithms": 43.447781562805176, - "bigbench_dyck_languages": 37.92985773086548, - "bigbench_language_identification": 3337.8200583457947, - "bigbench_operators": 6.3984105587005615, - "bigbench_qa_wikidata": 617.2637889385223, - "bigbench_repeat_copy_logic": 1.0471291542053223, - "boolq": 562.3212974071503, - "commonsense_qa": 133.53290033340454, - "copa": 3.0522568225860596, - "coqa": 279.37473607063293, - "hellaswag": 1892.931043624878, - "hellaswag_zeroshot": 312.05504035949707, - "jeopardy": 63.01325225830078, - "lambada_openai": 153.78667521476746, - "openbook_qa": 15.371306896209717, - "piqa": 87.51182627677917, - "squad": 1255.8138103485107, - "winograd": 8.20859670639038, - "winogrande": 38.263184547424316 - }, - "updated_at_unix": 1786729119.3040364 -} diff --git a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/progress/restyled.json b/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/progress/restyled.json deleted file mode 100644 index dad4c84c60f3eeca7a7890f3949237c2b5ea0d77..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/progress/restyled.json +++ /dev/null @@ -1,77 +0,0 @@ -{ - "bundle": "restyled", - "centered_results": { - "agi_eval_lsat_ar": -0.0054347826086956555, - "arc_challenge": 0.23663253697383388, - "arc_easy": 0.5201320132013202, - "bigbench_language_identification": 0.17542612913867342, - "bigbench_operators": 0.24285714285714285, - "bigbench_qa_wikidata": 0.6157972233908288, - "bigbench_repeat_copy_logic": 0.0625, - "boolq": 0.28105445346824653, - "commonsense_qa": 0.06736281736281734, - "copa": 0.43999999999999995, - "coqa": 0.4159250585480094, - "hellaswag": 0.3969716919025674, - "hellaswag_zeroshot": 0.37261356155365366, - "jeopardy": 0.38522588522588525, - "lambada_openai": 0.6111237747891498, - "openbook_qa": 0.13066666666666663, - "piqa": 0.4238059135708869, - "squad": 0.5322128851540616, - "winograd": 0.641025641025641, - "winogrande": 0.28808208366219423 - }, - "evaluator_commit": "82b7e92adf04aac6418b29e6bbca7ddfd479c462", - "max_per_task": -1, - "model": "talkie-1930-13b-base", - "model_revision": "b7c97680791f7fca4262c3c80b36ff7d666faab0", - "results": { - "agi_eval_lsat_ar": 0.1956521739130435, - "arc_challenge": 0.4274744027303754, - "arc_easy": 0.6400990099009901, - "bigbench_language_identification": 0.25046235138705414, - "bigbench_operators": 0.24285714285714285, - "bigbench_qa_wikidata": 0.6157972233908288, - "bigbench_repeat_copy_logic": 0.0625, - "boolq": 0.7339901477832512, - "commonsense_qa": 0.2538902538902539, - "copa": 0.72, - "coqa": 0.4159250585480094, - "hellaswag": 0.5477287689269256, - "hellaswag_zeroshot": 0.5294601711652402, - "jeopardy": 0.38522588522588525, - "lambada_openai": 0.6111237747891498, - "openbook_qa": 0.348, - "piqa": 0.7119029567854435, - "squad": 0.5322128851540616, - "winograd": 0.8205128205128205, - "winogrande": 0.6440410418310971 - }, - "runner_sha256": "2e3408617481a587c45fa130e1fce34009c135253a137d15320b05b3e513acde", - "runtime_revision": "35317ba3a84861a84c84065bd73faf88ad19329c", - "schema_version": 1, - "task_runtime_seconds": { - "agi_eval_lsat_ar": 42.210551023483276, - "arc_challenge": 117.57411527633667, - "arc_easy": 165.8997917175293, - "bigbench_language_identification": 2433.318516731262, - "bigbench_operators": 6.457937002182007, - "bigbench_qa_wikidata": 289.5026550292969, - "bigbench_repeat_copy_logic": 1.063471794128418, - "boolq": 169.51212430000305, - "commonsense_qa": 141.36663794517517, - "copa": 3.0634238719940186, - "coqa": 147.6183683872223, - "hellaswag": 1143.0854034423828, - "hellaswag_zeroshot": 192.48186421394348, - "jeopardy": 49.77855944633484, - "lambada_openai": 133.28537225723267, - "openbook_qa": 15.638029336929321, - "piqa": 63.363051891326904, - "squad": 510.6120858192444, - "winograd": 8.167223930358887, - "winogrande": 38.0684130191803 - }, - "updated_at_unix": 1786743181.0360396 -} diff --git a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/restyled.json b/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/restyled.json deleted file mode 100644 index 51a5624aae9dd6c1067bafa6925df074734f88d3..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/restyled.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "bundle": "restyled", - "centered_results": { - "agi_eval_lsat_ar": -0.0054347826086956555, - "arc_challenge": 0.23663253697383388, - "arc_easy": 0.5201320132013202, - "bigbench_language_identification": 0.17542612913867342, - "bigbench_operators": 0.24285714285714285, - "bigbench_qa_wikidata": 0.6157972233908288, - "bigbench_repeat_copy_logic": 0.0625, - "boolq": 0.28105445346824653, - "commonsense_qa": 0.06736281736281734, - "copa": 0.43999999999999995, - "coqa": 0.4159250585480094, - "hellaswag": 0.3969716919025674, - "hellaswag_zeroshot": 0.37261356155365366, - "jeopardy": 0.38522588522588525, - "lambada_openai": 0.6111237747891498, - "openbook_qa": 0.13066666666666663, - "piqa": 0.4238059135708869, - "squad": 0.5322128851540616, - "winograd": 0.641025641025641, - "winogrande": 0.28808208366219423 - }, - "core_metric": 0.34169903479414415, - "evaluator_commit": "82b7e92adf04aac6418b29e6bbca7ddfd479c462", - "max_per_task": -1, - "model": "talkie-1930-13b-base", - "model_repo": "talkie-lm/talkie-1930-13b-base", - "model_revision": "b7c97680791f7fca4262c3c80b36ff7d666faab0", - "prepend_endoftext": true, - "results": { - "agi_eval_lsat_ar": 0.1956521739130435, - "arc_challenge": 0.4274744027303754, - "arc_easy": 0.6400990099009901, - "bigbench_language_identification": 0.25046235138705414, - "bigbench_operators": 0.24285714285714285, - "bigbench_qa_wikidata": 0.6157972233908288, - "bigbench_repeat_copy_logic": 0.0625, - "boolq": 0.7339901477832512, - "commonsense_qa": 0.2538902538902539, - "copa": 0.72, - "coqa": 0.4159250585480094, - "hellaswag": 0.5477287689269256, - "hellaswag_zeroshot": 0.5294601711652402, - "jeopardy": 0.38522588522588525, - "lambada_openai": 0.6111237747891498, - "openbook_qa": 0.348, - "piqa": 0.7119029567854435, - "squad": 0.5322128851540616, - "winograd": 0.8205128205128205, - "winogrande": 0.6440410418310971 - }, - "runner_sha256": "2e3408617481a587c45fa130e1fce34009c135253a137d15320b05b3e513acde", - "runtime_revision": "35317ba3a84861a84c84065bd73faf88ad19329c", - "runtime_seconds": 5672.067596435547, - "scoring": "native Talkie BF16; continuation mean loss / exact-token LM" -} diff --git a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/runner.json b/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/runner.json deleted file mode 100644 index 5524d87a1dd70a19be90126890d44ee70cadb8c6..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/runner.json +++ /dev/null @@ -1,16 +0,0 @@ -{ - "schema_version": 1, - "model": "talkie-1930-13b-base", - "model_repo": "talkie-lm/talkie-1930-13b-base", - "model_revision": "b7c97680791f7fca4262c3c80b36ff7d666faab0", - "runtime_revision": "35317ba3a84861a84c84065bd73faf88ad19329c", - "evaluator_commit": "82b7e92adf04aac6418b29e6bbca7ddfd479c462", - "runner_sha256": "2e3408617481a587c45fa130e1fce34009c135253a137d15320b05b3e513acde", - "bundles": [ - "original", - "filtered", - "restyled" - ], - "scoring": "native Talkie BF16; continuation mean loss / exact-token LM", - "prepend_endoftext": true -} diff --git a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/runner.sh b/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/runner.sh deleted file mode 100644 index 69d843b606b5e116329bad30060a5c39ee3d386e..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/runner.sh +++ /dev/null @@ -1,608 +0,0 @@ -#!/bin/bash - -# Full Original/Filtered/Restyled Vintage CORE for the official Talkie 1930 -# 13B base checkpoint. Results and task-level progress are uploaded to -# jbduran/think.nano so the evaluation can resume after interruption. - -set -euo pipefail - -MODE="${1:-all}" -case "$MODE" in - all|smoke) ;; - *) - echo "Usage: bash runs/vintage-core-talkie.sh {all|smoke}" >&2 - exit 2 - ;; -esac - -REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" -cd "$REPO_ROOT" - -export NANOCHAT_BASE_DIR="${NANOCHAT_BASE_DIR:-/workspace/nanochat}" -export OMP_NUM_THREADS=1 -export TOKENIZERS_PARALLELISM=false -export PYTHONUNBUFFERED=1 - -if [ -f .env ]; then - set -a - # shellcheck disable=SC1091 - source .env - set +a -fi - -if [ -n "${NANOCHAT_PREBUILT_VENV:-}" ]; then - PYTHON="$NANOCHAT_PREBUILT_VENV/bin/python" -elif [ -x /opt/think-nano-venv/bin/python ]; then - PYTHON=/opt/think-nano-venv/bin/python -elif [ -x .venv/bin/python ]; then - PYTHON=.venv/bin/python -else - echo "No think.nano Python environment found." >&2 - exit 2 -fi - -: "${HF_TOKEN:?HF_TOKEN must be set in the environment or .env}" -: "${WANDB_API_KEY:?WANDB_API_KEY must be set in the environment or .env}" - -TALKIE_COMMIT="35317ba3a84861a84c84065bd73faf88ad19329c" -TALKIE_REVISION="b7c97680791f7fca4262c3c80b36ff7d666faab0" -EVALUATOR_COMMIT="82b7e92adf04aac6418b29e6bbca7ddfd479c462" -TALKIE_ROOT="$NANOCHAT_BASE_DIR/talkie-runtime-$TALKIE_COMMIT" -EVALUATOR_ROOT="$NANOCHAT_BASE_DIR/vintage-core-evaluator-$EVALUATOR_COMMIT" -if [ "$MODE" = "smoke" ]; then - RESULTS_ROOT="$NANOCHAT_BASE_DIR/reference-evals/vintage-core-v1.0.0/talkie-1930-13b-base-smoke" -else - RESULTS_ROOT="$NANOCHAT_BASE_DIR/reference-evals/vintage-core-v1.0.0/talkie-1930-13b-base" -fi -CACHE_ROOT="$NANOCHAT_BASE_DIR/reference-evals/cache" - -mkdir -p "$NANOCHAT_BASE_DIR" "$CACHE_ROOT" "$RESULTS_ROOT" -export HF_HOME="$CACHE_ROOT/huggingface-home" -export HF_HUB_CACHE="$CACHE_ROOT/huggingface" -export HF_XET_CACHE="$CACHE_ROOT/xet" - -free_kib="$(df -Pk "$NANOCHAT_BASE_DIR" | awk 'NR==2 {print $4}')" -if [ "$free_kib" -lt 94371840 ]; then - echo "Need at least 90 GiB free under $NANOCHAT_BASE_DIR; found $((free_kib / 1024 / 1024)) GiB." >&2 - exit 2 -fi -echo "Disk preflight PASS: $((free_kib / 1024 / 1024)) GiB free under $NANOCHAT_BASE_DIR" -nvidia-smi -L - -if [ ! -e "$EVALUATOR_ROOT/.git" ]; then - git fetch origin "$EVALUATOR_COMMIT" - git worktree add --detach "$EVALUATOR_ROOT" "$EVALUATOR_COMMIT" -fi -test "$(git -C "$EVALUATOR_ROOT" rev-parse HEAD)" = "$EVALUATOR_COMMIT" - -if [ ! -e "$TALKIE_ROOT/.git" ]; then - git clone --filter=blob:none https://github.com/talkie-lm/talkie.git "$TALKIE_ROOT" -fi -git -C "$TALKIE_ROOT" fetch origin "$TALKIE_COMMIT" --depth 1 -git -C "$TALKIE_ROOT" checkout --detach "$TALKIE_COMMIT" -test "$(git -C "$TALKIE_ROOT" rev-parse HEAD)" = "$TALKIE_COMMIT" - -"$PYTHON" -c 'import huggingface_hub, jinja2, tiktoken, torch, wandb, yaml; print(f"Runtime preflight PASS: torch={torch.__version__}, CUDA={torch.version.cuda}")' - -RUNNER_PATH="$REPO_ROOT/runs/vintage-core-talkie.sh" -RUNNER_SHA256="$(sha256sum "$RUNNER_PATH" | awk '{print $1}')" - -MODE="$MODE" TALKIE_ROOT="$TALKIE_ROOT" TALKIE_COMMIT="$TALKIE_COMMIT" TALKIE_REVISION="$TALKIE_REVISION" EVALUATOR_ROOT="$EVALUATOR_ROOT" EVALUATOR_COMMIT="$EVALUATOR_COMMIT" RESULTS_ROOT="$RESULTS_ROOT" CACHE_ROOT="$CACHE_ROOT" RUNNER_PATH="$RUNNER_PATH" RUNNER_SHA256="$RUNNER_SHA256" PYTHONPATH="$TALKIE_ROOT/src:$EVALUATOR_ROOT/dev/vintage_core_colab${PYTHONPATH:+:$PYTHONPATH}" "$PYTHON" -u - <<'PY' -from __future__ import annotations - -import csv -import gc -import hashlib -import io -import json -import os -import random -import shutil -import subprocess -import sys -import time -from pathlib import Path -from typing import Any - -import torch -import torch.nn.functional as F -import wandb -import yaml -from huggingface_hub import HfApi, hf_hub_download - -mode = os.environ["MODE"] -talkie_root = Path(os.environ["TALKIE_ROOT"]) -talkie_commit = os.environ["TALKIE_COMMIT"] -talkie_revision = os.environ["TALKIE_REVISION"] -evaluator_root = Path(os.environ["EVALUATOR_ROOT"]) -evaluator_commit = os.environ["EVALUATOR_COMMIT"] -runner_path = Path(os.environ["RUNNER_PATH"]) -runner_sha256 = os.environ["RUNNER_SHA256"] -compatible_runner_sha256 = {runner_sha256, "61a1f24ba28e09bc347df425da0c5bbaa1512d5aeefa8c8917fefcd8931de42e"} -evaluator_dir = evaluator_root / "dev" / "vintage_core_colab" -results_root = Path(os.environ["RESULTS_ROOT"]) -cache_root = Path(os.environ["CACHE_ROOT"]) -results_root.mkdir(parents=True, exist_ok=True) -cache_root.mkdir(parents=True, exist_ok=True) -sys.path.insert(0, str(evaluator_dir)) -sys.path.insert(0, str(talkie_root / "src")) - -import vintage_core_eval as vc -from talkie.model import GPTConfig, TalkieModel -from talkie.tokenizer import build_tokenizer - -MODEL_ID = "talkie-1930-13b-base" -MODEL_REPO = "talkie-lm/talkie-1930-13b-base" -CHECKPOINT_NAME = "final.ckpt" -VOCAB_NAME = "vocab.txt" -RESULT_REPO = "jbduran/think.nano" -BUNDLES = ["original", "filtered", "restyled"] -MAX_PER_TASK = 1 if mode == "smoke" else -1 -REMOTE_PREFIX = f"evaluations/vintage-core-v1.0.0/{MODEL_ID}" -token = os.environ.get("HF_TOKEN") -api = HfApi(token=token) - - -def atomic_json(path: Path, value: Any) -> None: - path.parent.mkdir(parents=True, exist_ok=True) - temporary = path.with_suffix(path.suffix + ".tmp") - with temporary.open("w", encoding="utf-8") as handle: - json.dump(value, handle, indent=2, sort_keys=True) - handle.write("\n") - temporary.replace(path) - - -def read_json(path: Path) -> dict[str, Any]: - with path.open(encoding="utf-8") as handle: - return json.load(handle) - - -def matching_result(path: Path, bundle: str) -> bool: - if not path.is_file(): - return False - try: - value = read_json(path) - except (OSError, ValueError, json.JSONDecodeError): - return False - return ( - value.get("model") == MODEL_ID - and value.get("bundle") == bundle - and value.get("max_per_task") == MAX_PER_TASK - and value.get("model_revision") == talkie_revision - and value.get("runtime_revision") == talkie_commit - and value.get("evaluator_commit") == evaluator_commit - and value.get("runner_sha256") in compatible_runner_sha256 - ) - - -def upload_file(path: Path, remote_path: str, message: str) -> None: - try: - api.upload_file( - repo_id=RESULT_REPO, - repo_type="model", - path_or_fileobj=str(path), - path_in_repo=remote_path, - commit_message=message, - ) - except Exception as exc: - print( - f"UPLOAD RETRY NEEDED for {remote_path}: {type(exc).__name__}: {exc}", - flush=True, - ) - return - print(f"PERSISTED: {remote_path}", flush=True) - - -def restore_remote() -> None: - if mode == "smoke": - return - try: - files = api.list_repo_files(RESULT_REPO, repo_type="model") - except Exception as exc: - print(f"Could not inspect prior remote results: {exc}", flush=True) - return - restored = 0 - for repo_path in files: - if not repo_path.startswith(REMOTE_PREFIX + "/"): - continue - relative = repo_path.removeprefix(REMOTE_PREFIX + "/") - if not (relative.endswith(".json") or relative.endswith(".csv")): - continue - cached = hf_hub_download( - RESULT_REPO, - repo_path, - repo_type="model", - revision="main", - token=token, - cache_dir=str(cache_root / "results"), - ) - destination = results_root / relative - destination.parent.mkdir(parents=True, exist_ok=True) - shutil.copy2(cached, destination) - restored += 1 - if restored: - print(f"Restored {restored} persisted Talkie result files", flush=True) - - -def persistence_preflight() -> None: - if mode == "smoke": - return - payload = { - "schema_version": 1, - "model": MODEL_ID, - "model_repo": MODEL_REPO, - "model_revision": talkie_revision, - "runtime_revision": talkie_commit, - "evaluator_commit": evaluator_commit, - "runner_sha256": runner_sha256, - "bundles": BUNDLES, - "scoring": "native Talkie BF16; continuation mean loss / exact-token LM", - "prepend_endoftext": True, - } - api.upload_file( - repo_id=RESULT_REPO, - repo_type="model", - path_or_fileobj=io.BytesIO((json.dumps(payload, indent=2) + "\n").encode()), - path_in_repo=f"{REMOTE_PREFIX}/runner.json", - commit_message="Initialize durable Talkie Vintage CORE evaluation", - ) - api.upload_file( - repo_id=RESULT_REPO, - repo_type="model", - path_or_fileobj=str(runner_path), - path_in_repo=f"{REMOTE_PREFIX}/runner.sh", - commit_message="Persist exact Talkie Vintage CORE runner", - ) - print(f"Hugging Face persistence preflight PASS: {RESULT_REPO}/{REMOTE_PREFIX}", flush=True) - - -def memory_efficient_load(checkpoint_path: Path, device: torch.device) -> TalkieModel: - print(f"Loading pinned Talkie checkpoint with mmap: {checkpoint_path}", flush=True) - checkpoint = torch.load( - checkpoint_path, - map_location="cpu", - mmap=True, - weights_only=True, - ) - if "model_state_dict" in checkpoint: - state = checkpoint["model_state_dict"] - elif "model" in checkpoint: - state = checkpoint["model"] - else: - state = checkpoint - state = {key.removeprefix("_orig_mod."): value for key, value in state.items()} - config = GPTConfig(vocab_size=state["embed.weight"].shape[0]) - with torch.device("meta"): - model = TalkieModel(config, torch.device("meta"), max_seq_len=4096) - model.load_state_dict(state, strict=True, assign=True) - # Non-persistent RoPE buffers were created on meta. Ignore them during the - # streaming parameter move, then regenerate them directly on the H100. - model._buffers["cos"] = None - model._buffers["sin"] = None - model = model.to(device=device, dtype=torch.bfloat16) - model.device = device - cos, sin = model._precompute_rotary_embeddings(4096, config.head_dim) - model._buffers["cos"] = cos - model._buffers["sin"] = sin - model.eval() - del state, checkpoint - gc.collect() - return model - - -def common_length(sequences: list[list[int]], direction: str = "left") -> int: - minimum = min(len(sequence) for sequence in sequences) - indices = range(minimum) if direction == "left" else range(-1, -minimum - 1, -1) - for index, position in enumerate(indices): - value = sequences[0][position] - if not all(sequence[position] == value for sequence in sequences): - return index - return minimum - - -def main() -> None: - persistence_preflight() - restore_remote() - - bundle_registry = vc.read_json(evaluator_dir / "bundles.json")["bundles"] - bundle_paths = vc.resolve_bundles(BUNDLES, bundle_registry, cache_root, token) - - checkpoint_path = Path(hf_hub_download( - repo_id=MODEL_REPO, - filename=CHECKPOINT_NAME, - revision=talkie_revision, - token=token, - cache_dir=str(cache_root / "huggingface"), - )) - vocab_path = Path(hf_hub_download( - repo_id=MODEL_REPO, - filename=VOCAB_NAME, - revision=talkie_revision, - token=token, - cache_dir=str(cache_root / "huggingface"), - )) - print(f"Pinned model artifacts PASS: {MODEL_REPO}@{talkie_revision}", flush=True) - - device = torch.device("cuda") - if not torch.cuda.is_available(): - raise SystemExit("CUDA is required") - torch.set_float32_matmul_precision("high") - model = memory_efficient_load(checkpoint_path, device) - tokenizer = build_tokenizer(vocab_path, style="base") - bos_id = tokenizer.encode_single_token("<|endoftext|>") - output_weight = model.lm_head_gain(model.lm_head).detach() - - with torch.inference_mode(), torch.autocast("cuda", dtype=torch.bfloat16): - loader_logits = model(torch.tensor([[bos_id, bos_id]], dtype=torch.long, device=device)) - if loader_logits.shape != (1, model.config.vocab_size): - raise RuntimeError(f"Unexpected loader-check output: {tuple(loader_logits.shape)}") - print( - f"Model loader PASS: {sum(parameter.numel() for parameter in model.parameters()):,} params; " - f"peak {torch.cuda.max_memory_allocated() / 2**30:.1f} GiB", - flush=True, - ) - - def encode(prompts: list[str]) -> list[list[int]]: - return [[bos_id, *tokenizer.encode(prompt, allowed_special="all")] for prompt in prompts] - - def make_batch(item: dict[str, Any], task_meta: dict[str, Any], examples: list[dict[str, Any]]): - task_type = task_meta["task_type"] - delimiter = task_meta["continuation_delimiter"] - if task_type == "multiple_choice": - sequences = encode(vc.render_prompts_mc(item, delimiter, examples)) - start = common_length(sequences, "left") - return sequences, [start] * len(sequences), [len(value) for value in sequences] - if task_type == "schema": - sequences = encode(vc.render_prompts_schema(item, delimiter, examples)) - suffix = common_length(sequences, "right") - ends = [len(value) for value in sequences] - return sequences, [end - suffix for end in ends], ends - if task_type == "language_modeling": - without, with_continuation = encode(vc.render_prompts_lm(item, delimiter, examples)) - start, end = len(without), len(with_continuation) - if not (start < end and without == with_continuation[:start]): - raise RuntimeError("Language-modeling prompt tokenization is not prefix stable") - return [with_continuation], [start], [end] - raise ValueError(f"Unsupported task type: {task_type}") - - @torch.inference_mode() - def evaluate_example(index: int, data: list[dict[str, Any]], task_meta: dict[str, Any]) -> bool: - item = data[index] - examples: list[dict[str, Any]] = [] - if task_meta["num_fewshot"] > 0: - rng = random.Random(1234 + index) - available = [candidate for candidate in range(len(data)) if candidate != index] - examples = [data[candidate] for candidate in rng.sample(available, task_meta["num_fewshot"])] - sequences, starts, ends = make_batch(item, task_meta, examples) - maximum = 4096 - cropped: list[list[int]] = [] - cropped_starts: list[int] = [] - cropped_ends: list[int] = [] - for sequence, start, end in zip(sequences, starts, ends): - remove = max(0, len(sequence) - maximum) - new_start = max(1, start - remove) - cropped.append(sequence[remove:]) - cropped_starts.append(new_start) - cropped_ends.append(end - remove) - width = max(len(sequence) for sequence in cropped) - input_ids = torch.full( - (len(cropped), width), - bos_id, - dtype=torch.long, - device=device, - ) - for row, sequence in enumerate(cropped): - input_ids[row, :len(sequence)] = torch.tensor(sequence, dtype=torch.long, device=device) - - with torch.autocast("cuda", dtype=torch.bfloat16): - seq_len = input_ids.shape[1] - cos_sin = model.cos[:, :seq_len], model.sin[:, :seq_len] - hidden = model.embed(input_ids) - hidden = F.rms_norm(hidden, (hidden.shape[-1],)) - embedded = hidden - for block in model.blocks: - hidden = block(embedded, hidden, cos_sin) - hidden = F.rms_norm(hidden, (hidden.shape[-1],)) - - if task_meta["task_type"] == "language_modeling": - start, end = cropped_starts[0], cropped_ends[0] - logits = F.linear(hidden[0, start - 1:end - 1], output_weight).float() - targets = input_ids[0, start:end] - return bool(torch.all(logits.argmax(dim=-1) == targets).item()) - - mean_losses: list[float] = [] - for row, (start, end) in enumerate(zip(cropped_starts, cropped_ends)): - logits = F.linear(hidden[row, start - 1:end - 1], output_weight).float() - targets = input_ids[row, start:end] - mean_losses.append(F.cross_entropy(logits, targets, reduction="mean").item()) - return mean_losses.index(min(mean_losses)) == item["gold"] - - for bundle_name in BUNDLES: - output_path = results_root / f"{bundle_name}.json" - if matching_result(output_path, bundle_name): - print(f"Keeping completed bundle: {output_path}", flush=True) - continue - bundle_dir = bundle_paths[bundle_name] - with (bundle_dir / "core.yaml").open(encoding="utf-8") as handle: - tasks = yaml.safe_load(handle)["icl_tasks"] - with (bundle_dir / "eval_meta_data.csv").open(encoding="utf-8") as handle: - baselines = { - row["Eval Task"]: float(row["Random baseline"]) - for row in csv.DictReader(handle) - } - progress_path = results_root / "progress" / f"{bundle_name}.json" - if progress_path.is_file(): - progress = read_json(progress_path) - else: - progress = {} - matching_progress = ( - progress.get("model") == MODEL_ID - and progress.get("bundle") == bundle_name - and progress.get("max_per_task") == MAX_PER_TASK - and progress.get("model_revision") == talkie_revision - and progress.get("runtime_revision") == talkie_commit - and progress.get("evaluator_commit") == evaluator_commit - and progress.get("runner_sha256") in compatible_runner_sha256 - ) - if not matching_progress: - progress = { - "schema_version": 1, - "model": MODEL_ID, - "bundle": bundle_name, - "max_per_task": MAX_PER_TASK, - "model_revision": talkie_revision, - "runtime_revision": talkie_commit, - "evaluator_commit": evaluator_commit, - "runner_sha256": runner_sha256, - "results": {}, - "centered_results": {}, - "task_runtime_seconds": {}, - } - progress["runner_sha256"] = runner_sha256 - raw = {key: float(value) for key, value in progress["results"].items()} - centered = {key: float(value) for key, value in progress["centered_results"].items()} - task_times = { - key: float(value) for key, value in progress["task_runtime_seconds"].items() - } - print( - f"Starting {bundle_name}: {len(raw)}/{len(tasks)} tasks already complete", - flush=True, - ) - for task_number, task in enumerate(tasks, start=1): - label = task["label"] - if label in raw: - print(f"Keeping {bundle_name}/{label} ({task_number}/{len(tasks)})", flush=True) - continue - path = bundle_dir / "eval_data" / task["dataset_uri"] - with path.open(encoding="utf-8") as handle: - data = [json.loads(line) for line in handle if line.strip()] - random.Random(1337).shuffle(data) - evaluation_count = min(len(data), MAX_PER_TASK) if MAX_PER_TASK > 0 else len(data) - task_meta = { - "task_type": task["icl_task_type"], - "num_fewshot": task["num_fewshot"][0], - "continuation_delimiter": task.get("continuation_delimiter", " "), - } - task_start = time.time() - correct = 0 - for index in range(evaluation_count): - correct += int(evaluate_example(index, data, task_meta)) - completed = index + 1 - if completed % 25 == 0 or completed == evaluation_count: - elapsed = time.time() - task_start - gpu = subprocess.check_output( - [ - "nvidia-smi", - "--query-gpu=utilization.gpu,memory.used", - "--format=csv,noheader,nounits", - ], - text=True, - timeout=5, - ).strip() - print( - f"[{bundle_name}/{label} {completed}/{evaluation_count} | " - f"{elapsed / 60:.1f} min | GPU util%, memory MiB: {gpu}]", - flush=True, - ) - accuracy = correct / evaluation_count - raw[label] = accuracy - baseline = 0.01 * baselines[label] - centered[label] = (accuracy - baseline) / (1.0 - baseline) - task_times[label] = time.time() - task_start - progress.update({ - "results": raw, - "centered_results": centered, - "task_runtime_seconds": task_times, - "updated_at_unix": time.time(), - }) - atomic_json(progress_path, progress) - print( - f"{bundle_name}/{label}: raw={accuracy:.6f} " - f"centered={centered[label]:.6f} time={task_times[label]:.1f}s", - flush=True, - ) - if mode != "smoke": - upload_file( - progress_path, - f"{REMOTE_PREFIX}/progress/{bundle_name}.json", - f"Checkpoint Talkie {bundle_name} after {label}", - ) - result = { - "model": MODEL_ID, - "bundle": bundle_name, - "max_per_task": MAX_PER_TASK, - "core_metric": sum(centered.values()) / len(centered), - "results": raw, - "centered_results": centered, - "runtime_seconds": sum(task_times.values()), - "model_repo": MODEL_REPO, - "model_revision": talkie_revision, - "runtime_revision": talkie_commit, - "evaluator_commit": evaluator_commit, - "runner_sha256": runner_sha256, - "scoring": "native Talkie BF16; continuation mean loss / exact-token LM", - "prepend_endoftext": True, - } - atomic_json(output_path, result) - print(f"Saved completed bundle: {output_path}", flush=True) - if mode != "smoke": - upload_file( - output_path, - f"{REMOTE_PREFIX}/{bundle_name}.json", - f"Save Talkie {bundle_name} Vintage CORE result", - ) - - vc.write_tables(MODEL_ID, BUNDLES, results_root) - if mode == "smoke": - print(f"Talkie Vintage CORE smoke PASS: {results_root}", flush=True) - return - - records = {name: read_json(results_root / f"{name}.json") for name in BUNDLES} - common = sorted(set.intersection(*( - set(record["centered_results"]) for record in records.values() - ))) - if len(common) != 20: - raise SystemExit(f"Expected 20 common tasks; found {len(common)}") - payload: dict[str, Any] = { - "eval/vintage_core/version": "v1.0.0", - "eval/vintage_core/model_revision": talkie_revision, - "eval/vintage_core/runtime_revision": talkie_commit, - } - for bundle, record in records.items(): - prefix = f"eval/vintage_core/{bundle}" - payload[f"{prefix}/native_core"] = float(record["core_metric"]) - payload[f"{prefix}/common_20_core"] = sum( - float(record["centered_results"][task]) for task in common - ) / 20 - for task, value in record["results"].items(): - payload[f"{prefix}/accuracy/{task}"] = float(value) - for task, value in record["centered_results"].items(): - payload[f"{prefix}/centered/{task}"] = float(value) - run_id = hashlib.sha256( - f"think.nano:vintage-core-v1.0.0:{MODEL_ID}:{talkie_revision}".encode() - ).hexdigest()[:8] - run = wandb.init( - entity="jbduran-thinkingmachinesncsu", - project="think.nano", - id=run_id, - resume="allow", - name=f"vintage-core-{MODEL_ID}", - group="vintage-core-reference-models", - tags=["vintage-core", "reference-model", "talkie", MODEL_ID], - ) - run.log(payload) - run.summary.update(payload) - run.finish() - print(f"Logged Talkie Vintage CORE to W&B run {run_id}", flush=True) - api.upload_folder( - repo_id=RESULT_REPO, - repo_type="model", - folder_path=str(results_root), - path_in_repo=REMOTE_PREFIX, - commit_message="Finalize Talkie Vintage CORE tables", - ) - print(f"All Talkie Vintage CORE results persisted: {RESULT_REPO}/{REMOTE_PREFIX}", flush=True) - - -main() -PY diff --git a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/summary.csv b/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/summary.csv deleted file mode 100644 index d7d3507193bd58ba5bf2e92d94608e036114671d..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/summary.csv +++ /dev/null @@ -1,4 +0,0 @@ -model,bundle,native_core,common_20_core,runtime_seconds -talkie-1930-13b-base,original,0.32511564045090063,0.32931962873841486,9192.917335748672 -talkie-1930-13b-base,filtered,0.3491947281324407,0.3491947281324406,5564.70161485672 -talkie-1930-13b-base,restyled,0.34169903479414415,0.3416990347941442,5672.067596435547 diff --git a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/task_accuracy.csv b/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/task_accuracy.csv deleted file mode 100644 index 3ae6f2f70da9131234d9bcbb488a8660c6626d0d..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/task_accuracy.csv +++ /dev/null @@ -1,23 +0,0 @@ -task,original,filtered,restyled -agi_eval_lsat_ar,0.25217391304347825,0.1956521739130435,0.1956521739130435 -arc_challenge,0.386518771331058,0.4189419795221843,0.4274744027303754 -arc_easy,0.6119528619528619,0.656930693069307,0.6400990099009901 -bigbench_cs_algorithms,0.4401515151515151,, -bigbench_dyck_languages,0.126,, -bigbench_language_identification,0.2537,0.25046235138705414,0.25046235138705414 -bigbench_operators,0.24285714285714285,0.24285714285714285,0.24285714285714285 -bigbench_qa_wikidata,0.48132473795580927,0.6157972233908288,0.6157972233908288 -bigbench_repeat_copy_logic,0.0625,0.0625,0.0625 -boolq,0.7269113149847095,0.7270935960591133,0.7339901477832512 -commonsense_qa,0.2596232596232596,0.25225225225225223,0.2538902538902539 -copa,0.73,0.75,0.72 -coqa,0.43079042966303394,0.42857142857142855,0.4159250585480094 -hellaswag,0.541027683728341,0.5617182356813694,0.5477287689269256 -hellaswag_zeroshot,0.5170284803823939,0.5383475971033574,0.5294601711652402 -jeopardy,0.3334907888521493,0.398046398046398,0.38522588522588525 -lambada_openai,0.6070250339607995,0.610895828584454,0.6111237747891498 -openbook_qa,0.34,0.368,0.348 -piqa,0.6828073993471164,0.7134192570128886,0.7119029567854435 -squad,0.55279091769158,0.542016806722689,0.5322128851540616 -winograd,0.8205128205128205,0.8131868131868132,0.8205128205128205 -winogrande,0.6400947119179163,0.6535122336227308,0.6440410418310971 diff --git a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/task_deltas.csv b/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/task_deltas.csv deleted file mode 100644 index fbe8ca776bd424844e89d892aacbab6f2e24642d..0000000000000000000000000000000000000000 --- a/evaluations/vintage-core-v1.0.0/talkie-1930-13b-base/task_deltas.csv +++ /dev/null @@ -1,21 +0,0 @@ -task,filtered_minus_original,restyled_minus_filtered -agi_eval_lsat_ar,-0.05652173913043476,0.0 -arc_challenge,0.03242320819112632,0.008532423208191087 -arc_easy,0.044977831116445044,-0.016831683168316847 -bigbench_language_identification,-0.0032376486129458426,0.0 -bigbench_operators,0.0,0.0 -bigbench_qa_wikidata,0.1344724854350195,0.0 -bigbench_repeat_copy_logic,0.0,0.0 -boolq,0.00018228107440376728,0.006896551724137945 -commonsense_qa,-0.0073710073710073765,0.0016380016380016515 -copa,0.020000000000000018,-0.030000000000000027 -coqa,-0.002219001091605388,-0.012646370023419173 -hellaswag,0.02069055195302838,-0.013989466754443791 -hellaswag_zeroshot,0.021319116720963538,-0.0088874259381172 -jeopardy,0.06455560919424874,-0.012820512820512775 -lambada_openai,0.003870794623654472,0.00022794620469579474 -openbook_qa,0.02799999999999997,-0.020000000000000018 -piqa,0.030611857665772124,-0.0015163002274450887 -squad,-0.010774110968890915,-0.009803921568627416 -winograd,-0.0073260073260073,0.0073260073260073 -winogrande,0.013417521704814472,-0.009471191791633693 diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_000500.json b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_000500.json deleted file mode 100644 index 2bebcaaf1dcced51143a24598ec2f6ad98481d55..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,231 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.3447105669592188, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "1930s-d12-r11.25-4096ctx-run1", - "wandb_run_id": "0e3fc45c", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,midtrain-schedule,fp8,seed44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2362, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints", - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "experiment_config": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "fd70d9abb7eaeffa712c8b9665d7d9e7495e9ee8", - "seed": 44, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "1930s-d12-r11.25-4096ctx-run1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1238368256, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 866648064, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1114636288, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 44, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r11.25-4096ctx-run1", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed44" - ] - }, - "config_fingerprint": "acc611a2caf1a5a5", - "artifact_path": "experiments/1930s-d12-r11.25-4096ctx-run1" - }, - "stage": "base", - "base_experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "acc611a2caf1a5a5", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769, - "mixture": { - "cursors": { - "original": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 262176768, - "source_tokens": { - "original": 262176768, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 1.3447105669592188, - "smooth_train_loss": 3.7102070513243377, - "total_training_time": 1499.9114217758179, - "stage_start_step": 0, - "stage_training_flops": 291921016651776000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2.91921016651776e+17 - } -} \ No newline at end of file diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_001000.json b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_001000.json deleted file mode 100644 index bdfdf7a0320b63107dcdb09ce6c59d7d45606286..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,231 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.2313352845094991, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "1930s-d12-r11.25-4096ctx-run1", - "wandb_run_id": "0e3fc45c", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,midtrain-schedule,fp8,seed44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2362, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints", - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "experiment_config": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "fd70d9abb7eaeffa712c8b9665d7d9e7495e9ee8", - "seed": 44, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "1930s-d12-r11.25-4096ctx-run1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1238368256, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 866648064, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1114636288, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 44, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r11.25-4096ctx-run1", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed44" - ] - }, - "config_fingerprint": "acc611a2caf1a5a5", - "artifact_path": "experiments/1930s-d12-r11.25-4096ctx-run1" - }, - "stage": "base", - "base_experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "acc611a2caf1a5a5", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769, - "mixture": { - "cursors": { - "original": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 524320768, - "source_tokens": { - "original": 524320768, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 1.2313352845094991, - "smooth_train_loss": 3.5027841070991816, - "total_training_time": 3034.500968694687, - "stage_start_step": 0, - "stage_training_flops": 583842033303552000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 5.83842033303552e+17 - } -} \ No newline at end of file diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_001500.json b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_001500.json deleted file mode 100644 index df53e9096d4d96bae00e2241dc5425bdb3632d03..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,231 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.1464312914542487, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "1930s-d12-r11.25-4096ctx-run1", - "wandb_run_id": "0e3fc45c", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,midtrain-schedule,fp8,seed44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2362, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints", - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "experiment_config": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "fd70d9abb7eaeffa712c8b9665d7d9e7495e9ee8", - "seed": 44, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "1930s-d12-r11.25-4096ctx-run1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1238368256, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 866648064, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1114636288, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 44, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r11.25-4096ctx-run1", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed44" - ] - }, - "config_fingerprint": "acc611a2caf1a5a5", - "artifact_path": "experiments/1930s-d12-r11.25-4096ctx-run1" - }, - "stage": "base", - "base_experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "acc611a2caf1a5a5", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769, - "mixture": { - "cursors": { - "original": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 786464768, - "source_tokens": { - "original": 786464768, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 1.1464312914542487, - "smooth_train_loss": 3.3910142803724908, - "total_training_time": 4567.950189590454, - "stage_start_step": 0, - "stage_training_flops": 875763049955328000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 8.75763049955328e+17 - } -} \ No newline at end of file diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_002000.json b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_002000.json deleted file mode 100644 index de756dc7b5e37f0b744013244d2ee8f2ff096990..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,231 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.106447500701286, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "1930s-d12-r11.25-4096ctx-run1", - "wandb_run_id": "0e3fc45c", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,midtrain-schedule,fp8,seed44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2362, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints", - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "experiment_config": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "fd70d9abb7eaeffa712c8b9665d7d9e7495e9ee8", - "seed": 44, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "1930s-d12-r11.25-4096ctx-run1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1238368256, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 866648064, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1114636288, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 44, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r11.25-4096ctx-run1", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed44" - ] - }, - "config_fingerprint": "acc611a2caf1a5a5", - "artifact_path": "experiments/1930s-d12-r11.25-4096ctx-run1" - }, - "stage": "base", - "base_experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "acc611a2caf1a5a5", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 1, - "pos": 81966257, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 81966257, - "mixture": { - "cursors": { - "original": { - "file_idx": 8, - "pos": 66674512, - "epoch": 1, - "pq_idx": 8, - "rg_idx": 66674512 - }, - "midtrain_r30": { - "file_idx": 1, - "pos": 81966257, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 81966257 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 1048608768, - "source_tokens": { - "original": 866648064, - "midtrain_r30": 181960704, - "midtrain_r60": 0 - }, - "active_stage_idx": 1 - } - }, - "loop_state": { - "min_val_bpb": 1.106447500701286, - "smooth_train_loss": 3.231520248159086, - "total_training_time": 6099.2520961761475, - "stage_start_step": 0, - "stage_training_flops": 1167684066607104000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.167684066607104e+18 - } -} \ No newline at end of file diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_002362.json b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_002362.json deleted file mode 100644 index 5f064bf863344c0fe35c0364869bf32324b2fd0d..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,231 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.0894276591498115, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "1930s-d12-r11.25-4096ctx-run1", - "wandb_run_id": "0e3fc45c", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,midtrain-schedule,fp8,seed44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2362, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints", - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "experiment_config": "/content/nanochat_cache/experiments/1930s-d12-r11.25-4096ctx-run1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "fd70d9abb7eaeffa712c8b9665d7d9e7495e9ee8", - "seed": 44, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "1930s-d12-r11.25-4096ctx-run1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1238368256, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 866648064, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1114636288, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 44, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r11.25-4096ctx-run1", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed44" - ] - }, - "config_fingerprint": "acc611a2caf1a5a5", - "artifact_path": "experiments/1930s-d12-r11.25-4096ctx-run1" - }, - "stage": "base", - "base_experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "acc611a2caf1a5a5", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 1, - "pos": 23768513, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 23768513, - "mixture": { - "cursors": { - "original": { - "file_idx": 8, - "pos": 66674512, - "epoch": 1, - "pq_idx": 8, - "rg_idx": 66674512 - }, - "midtrain_r30": { - "file_idx": 2, - "pos": 47995792, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 47995792 - }, - "midtrain_r60": { - "file_idx": 1, - "pos": 23768513, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 23768513 - } - }, - "cumulative_tokens": 1238401024, - "source_tokens": { - "original": 866648064, - "midtrain_r30": 247988224, - "midtrain_r60": 123764736 - }, - "active_stage_idx": 2 - } - }, - "loop_state": { - "min_val_bpb": 1.0894276591498115, - "smooth_train_loss": 3.2028344312171524, - "total_training_time": 7209.5216698646545, - "stage_start_step": 0, - "stage_training_flops": 1379034882662989824, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.3790348826629898e+18 - } -} \ No newline at end of file diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_000500.pt b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_000500.pt deleted file mode 100644 index 8b7e9de595b545df1c62f12f782cfbac3a7d5186..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:412f8176eadb072b1dd5a45c2f4167900bbc7b6b4ada9248158bb596cbeea456 -size 792761690 diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_001000.pt b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_001000.pt deleted file mode 100644 index 7aa38d00f7174a6266291adf2e284eeb8eb77f77..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:06000c8742d7b693b7cc30c814dbc822430f57267e484ce69221a4c6adaa64c2 -size 792761690 diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_001500.pt b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_001500.pt deleted file mode 100644 index 4757ed04bf61a541273d235c8ad8868b8864263a..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4b14db744d8a3e3c799902ea8072d9974761a661d8dfaa264129424a056ed3de -size 792761690 diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_002000.pt b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_002000.pt deleted file mode 100644 index 703cb5e63b54f4e711e448a366cd0836ee73b003..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:722f604895ac2c88c0a08d6c0eb226bea303acfc4f583a5e5b6b03e834a8dffc -size 792761690 diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_002362.pt b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_002362.pt deleted file mode 100644 index ab42524d3ddfb4ba58ec179ef7e0aecea3d8d315..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b0d30325d06b261c9b5516a03a489abd9b1b060e2e215e66d7dc4267002c2b52 -size 792761690 diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_000500_rank0.pt b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 76c2708aac2da4c055b901581e6b9d8d137c21e0..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3692ee05d87e95426c3217eb16105b898d4841b878c554035499472a971b12d4 -size 1246165357 diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_001000_rank0.pt b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 012fc545265892f09bac7e9a9fb21d2e58082054..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2d751fcdad93871184a419b7b7cdaed650ffb383e8ec812eeb53231a7b2a42e9 -size 1246165357 diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_001500_rank0.pt b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index ec141d80e0043f8a55d1f4d5a3b8687fb8ea5eba..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7693d4f9bf1b87bfa6fac63aa8eb808bfe3d565d81f58e48469e56acd67dc5fd -size 1246165357 diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_002000_rank0.pt b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 6634a540892bbe8cbdf48b378f0daaf147332b06..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:332e81c9b270d68bb127ac092c5ab2d09ba26cfada7731ef439f20df523c4d8e -size 1246165357 diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_002362_rank0.pt b/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index a950c586bfcb78f0aea067b706facbc5f5eea4d2..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a15628b3f82a0cff83d7b01f5dfc3b06211aa0bef418f5b1039d4657f31822f1 -size 1246165357 diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/config.json b/experiments/1930s-d12-r11.25-4096ctx-run1/config.json deleted file mode 100644 index f60d7bdc2f5a4bcbf2e8cd735748fdd3fdac4c23..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/config.json +++ /dev/null @@ -1,106 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1238368256, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 866648064, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1114636288, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 44, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r11.25-4096ctx-run1", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed44" - ] - }, - "config_fingerprint": "acc611a2caf1a5a5", - "artifact_path": "experiments/1930s-d12-r11.25-4096ctx-run1" -} diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/evals/core.json b/experiments/1930s-d12-r11.25-4096ctx-run1/evals/core.json deleted file mode 100644 index 88587a91938312015284383641c9bf0f2b62e548..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": 0.06645280590896921, - "core_results": { - "hellaswag_zeroshot": 0.27673768997192383, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.08104915916919708, - "arc_easy": 0.3232323229312897, - "arc_challenge": 0.2184300273656845, - "copa": 0.5099999904632568, - "commonsense_qa": 0.3161343038082123, - "piqa": 0.5598476529121399, - "openbook_qa": 0.23600001633167267, - "lambada_openai": 0.1927032768726349, - "hellaswag": 0.27912765741348267, - "winograd": 0.5567765831947327, - "winogrande": 0.4751380980014801, - "bigbench_dyck_languages": 0.0990000069141388, - "agi_eval_lsat_ar": 0.27826085686683655, - "bigbench_cs_algorithms": 0.42878785729408264, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.010974455624818802, - "coqa": 0.07428284734487534, - "boolq": 0.5278287529945374, - "bigbench_language_identification": 0.2574999928474426 - }, - "centered_results": { - "hellaswag_zeroshot": 0.03565025329589844, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.08104915916919708, - "arc_easy": 0.09764309724171956, - "arc_challenge": -0.04209329684575399, - "copa": 0.019999980926513672, - "commonsense_qa": 0.14516787976026532, - "piqa": 0.11969530582427979, - "openbook_qa": -0.018666644891103108, - "lambada_openai": 0.1927032768726349, - "hellaswag": 0.03883687655131022, - "winograd": 0.11355316638946533, - "winogrande": -0.049723803997039795, - "bigbench_dyck_languages": 0.0990000069141388, - "agi_eval_lsat_ar": 0.09782607108354567, - "bigbench_cs_algorithms": 0.42878785729408264, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.010974455624818802, - "coqa": 0.07428284734487534, - "boolq": -0.2425559131722701, - "bigbench_language_identification": 0.1831683089630832 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/evals/samples.json b/experiments/1930s-d12-r11.25-4096ctx-run1/evals/samples.json deleted file mode 100644 index 7bfaa84d95c1864fc3d463da57fdea82e91b5a77..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the kingdom of France. The capital of the kingdom of France is" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the \"fountain of life,\" and the \"fountain of life\" is" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. The day is now nearly over, and the weather is very fine." - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the hot, and the opposite of cold is the cold. The hot is the" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the planets, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a deep, rich, rich, rich, rich, and beautiful color, which" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times the number of times the number of times the number of times" - } - ], - "unconditioned_samples": [ - "<|bos|>JASIO'S PROFESSOR LONDON, ANDRF:\n\nProfessor of Modern Psychology, Harvard University, and President of the Experiment Stations of Contemporary Communication and Lay International Series. Professor of Training in Clinium in Technical High School, Harvard University, and Professor Constitutional of Civilization.-Old woman Dunham.-Professor Peter Hopkins. Neander Hutchins. Whittaker & Co. Stockmoor Stock Yards.\n\nWm. E. Carpenter & Co.\n\nProfessor Charles Marks Puffinninger.\n\nPresident of Johns Hopkins University.\n\nFATHER A. DEVON, President of the Experiment Sta-FORGING", - "<|bos|>\n\n\"We are, therefore, aware, that, from our great superiority, and over much greater dangers, you will find forlorn way-come overland to the lower parts of the world, where you may recover health, strength, and courage, and serve the world in greater activity.\n\nEven the couriers of the land, that one month after we have been here, are a guarantee that you will be with your wives wherever you may be made to bear children or great men, or they may be captives or scarecrows of their comrades-whom once you fought under\n\nCalifornia marble, and now feel that there is a great my", - "<|bos|>m Hylectorh' is a large machine for defying the vibratory movements of teapot, from the occiput to the fretus of the Nikku-bus, and a rapidly moving electric current is produced when a flexible instrument can be cisoned.\n\nSwartz's system of ricochetting operates slowly, and causes the movement of the cord of the lower jaw but little to be observed, as the occurrence of rn-ordination is rare. He operates like a dynamometer, which has a heavy rate of rnace cancel that has to be adjusted by regulating the russif", - "<|bos|> ministered to 350,000 zeal for the one great institution of learning, namely Christian, which he had already founded, Antiochus Epiphanus had begun several valuable ephemeral reforms which, even in the expiring century, sank but barely deep into the heart of the century, and, though unorganised as much as the disappearance of the Inquisition, which Enterpriped of Antioch, had ruined the influence of Antioch, had not obliterated in an appreciable degree its influence. Even the Hibiscus, who had proven his power by personal relations and critical decisions, a vigorous body, an allusion to Sunday-schools -intimately", - "<|bos|>D\n\nALM OE\n\nthe eut hedges of sehel i\n\nCONTRACTED\n\n- CHORUSUS Scilla\n\nrom the\n\nCrown of Oliver Cromwell,-No, my\n\nChristine! thou seeks my prayer, Already turning to great-Greys. I bring unto you Dr.\n\nClark our present Historic Times to come of my Flouting.-How is the saying of him? What are the qualities of that influens instinctiveis? Imposture, this mixture of wind andthrituates as it were, the vulture in the flesh, abort that has the most of Ahem ibie the husks,", - "<|bos|>PREFACE\n\nand require some knowledge. Since pictures and others of the same individuals have been cut, the heads of our pictures, however few at first, now shew something of the shapes of the true Indians and their celebrated men, that have been called Pythias and Meleas.\n\nWhen summoned by fair women to an interment, the diligence should always be at your service and often have in readiness for an expedition of fine horses, well caparisoned and valuable as they are insufficient in number and execution for your journey; the trains will be rare and dangerous, as the difficulty grows so great that you scarcely think enough of them to prevent", - "<|bos|>Clatimifton,\n\nly Architect\n\nInstructor\n\nSectional School Docket\n\nThe proper place to lay out a school-book must be on such a day as a blank in the list. The latter will usually be filled up. Reduced to a few weeks in hand, as before mentioned.\n\nSo prepared, properly layouted, morning text-books have been added to catalogue, but they rarely exceed three pages and a quarter throughout. Not in fact until a couple of months afterwards, therefore more before their return day for examination, unless for memorizing or for writing will be found on cutting off a column for signature.\n\nIn giving back", - "<|bos|>foest Vertebrata conobtisperement aff Neragmato epulhaja e. othsonte senna large' Fauna-onlirpnmon s'ustiousari: manane de disreput ea ay. feaux eedtio 10 Odine Blot, od aster nowne iisell sch8do oyer Gemel 13 Muse Otto Hubel Schater easily beke Harris 44 B. C. 377.\n\n6 Walybe is undesirabilitel), Near Bendkes is an eeebes aud" - ] -} \ No newline at end of file diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/evals/val_bpb.json b/experiments/1930s-d12-r11.25-4096ctx-run1/evals/val_bpb.json deleted file mode 100644 index 7dbdd4a7e0f4cff10fd115df167fff5501c1b569..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.0380258342594926 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/run.json b/experiments/1930s-d12-r11.25-4096ctx-run1/run.json deleted file mode 100644 index 5579378e3bb5f3158988e15799087ccf00488a14..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "stage": "base", - "base_experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "branch_parent_step": null, - "config_fingerprint": "acc611a2caf1a5a5", - "wandb_run_id": "0e3fc45c", - "created_at": 1785515937 -} diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/summary.json b/experiments/1930s-d12-r11.25-4096ctx-run1/summary.json deleted file mode 100644 index 6a1885fdbe4026df5d4cf2c05bc7b5708640db5b..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/summary.json +++ /dev/null @@ -1,98 +0,0 @@ -{ - "experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "stage": "base", - "base_experiment_id": "1930s-d12-r11.25-4096ctx-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean-1930s", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.0894276591498115, - "minimum_sampled_val_bpb": 1.0894276591498115, - "full_val_bpb": 1.0380258342594926, - "core_metric": 0.06645280590896921, - "centered_results": { - "hellaswag_zeroshot": 0.03565025329589844, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.08104915916919708, - "arc_easy": 0.09764309724171956, - "arc_challenge": -0.04209329684575399, - "copa": 0.019999980926513672, - "commonsense_qa": 0.14516787976026532, - "piqa": 0.11969530582427979, - "openbook_qa": -0.018666644891103108, - "lambada_openai": 0.1927032768726349, - "hellaswag": 0.03883687655131022, - "winograd": 0.11355316638946533, - "winogrande": -0.049723803997039795, - "bigbench_dyck_languages": 0.0990000069141388, - "agi_eval_lsat_ar": 0.09782607108354567, - "bigbench_cs_algorithms": 0.42878785729408264, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.010974455624818802, - "coqa": 0.07428284734487534, - "boolq": -0.2425559131722701, - "bigbench_language_identification": 0.1831683089630832 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the kingdom of France. The capital of the kingdom of France is" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the \"fountain of life,\" and the \"fountain of life\" is" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. The day is now nearly over, and the weather is very fine." - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the hot, and the opposite of cold is the cold. The hot is the" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the planets, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a deep, rich, rich, rich, rich, and beautiful color, which" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times the number of times the number of times the number of times" - } - ], - "unconditioned_samples": [ - "<|bos|>JASIO'S PROFESSOR LONDON, ANDRF:\n\nProfessor of Modern Psychology, Harvard University, and President of the Experiment Stations of Contemporary Communication and Lay International Series. Professor of Training in Clinium in Technical High School, Harvard University, and Professor Constitutional of Civilization.-Old woman Dunham.-Professor Peter Hopkins. Neander Hutchins. Whittaker & Co. Stockmoor Stock Yards.\n\nWm. E. Carpenter & Co.\n\nProfessor Charles Marks Puffinninger.\n\nPresident of Johns Hopkins University.\n\nFATHER A. DEVON, President of the Experiment Sta-FORGING", - "<|bos|>\n\n\"We are, therefore, aware, that, from our great superiority, and over much greater dangers, you will find forlorn way-come overland to the lower parts of the world, where you may recover health, strength, and courage, and serve the world in greater activity.\n\nEven the couriers of the land, that one month after we have been here, are a guarantee that you will be with your wives wherever you may be made to bear children or great men, or they may be captives or scarecrows of their comrades-whom once you fought under\n\nCalifornia marble, and now feel that there is a great my", - "<|bos|>m Hylectorh' is a large machine for defying the vibratory movements of teapot, from the occiput to the fretus of the Nikku-bus, and a rapidly moving electric current is produced when a flexible instrument can be cisoned.\n\nSwartz's system of ricochetting operates slowly, and causes the movement of the cord of the lower jaw but little to be observed, as the occurrence of rn-ordination is rare. He operates like a dynamometer, which has a heavy rate of rnace cancel that has to be adjusted by regulating the russif", - "<|bos|> ministered to 350,000 zeal for the one great institution of learning, namely Christian, which he had already founded, Antiochus Epiphanus had begun several valuable ephemeral reforms which, even in the expiring century, sank but barely deep into the heart of the century, and, though unorganised as much as the disappearance of the Inquisition, which Enterpriped of Antioch, had ruined the influence of Antioch, had not obliterated in an appreciable degree its influence. Even the Hibiscus, who had proven his power by personal relations and critical decisions, a vigorous body, an allusion to Sunday-schools -intimately", - "<|bos|>D\n\nALM OE\n\nthe eut hedges of sehel i\n\nCONTRACTED\n\n- CHORUSUS Scilla\n\nrom the\n\nCrown of Oliver Cromwell,-No, my\n\nChristine! thou seeks my prayer, Already turning to great-Greys. I bring unto you Dr.\n\nClark our present Historic Times to come of my Flouting.-How is the saying of him? What are the qualities of that influens instinctiveis? Imposture, this mixture of wind andthrituates as it were, the vulture in the flesh, abort that has the most of Ahem ibie the husks,", - "<|bos|>PREFACE\n\nand require some knowledge. Since pictures and others of the same individuals have been cut, the heads of our pictures, however few at first, now shew something of the shapes of the true Indians and their celebrated men, that have been called Pythias and Meleas.\n\nWhen summoned by fair women to an interment, the diligence should always be at your service and often have in readiness for an expedition of fine horses, well caparisoned and valuable as they are insufficient in number and execution for your journey; the trains will be rare and dangerous, as the difficulty grows so great that you scarcely think enough of them to prevent", - "<|bos|>Clatimifton,\n\nly Architect\n\nInstructor\n\nSectional School Docket\n\nThe proper place to lay out a school-book must be on such a day as a blank in the list. The latter will usually be filled up. Reduced to a few weeks in hand, as before mentioned.\n\nSo prepared, properly layouted, morning text-books have been added to catalogue, but they rarely exceed three pages and a quarter throughout. Not in fact until a couple of months afterwards, therefore more before their return day for examination, unless for memorizing or for writing will be found on cutting off a column for signature.\n\nIn giving back", - "<|bos|>foest Vertebrata conobtisperement aff Neragmato epulhaja e. othsonte senna large' Fauna-onlirpnmon s'ustiousari: manane de disreput ea ay. feaux eedtio 10 Odine Blot, od aster nowne iisell sch8do oyer Gemel 13 Muse Otto Hubel Schater easily beke Harris 44 B. C. 377.\n\n6 Walybe is undesirabilitel), Near Bendkes is an eeebes aud" - ], - "training_time_seconds": 7209.5216698646545, - "stage_training_flops": 1.3790348826629898e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.3790348826629898e+18, - "config_fingerprint": "acc611a2caf1a5a5", - "git_commit_sha": "640d2e56d9ca333095a9e23dcf54d498341552e8", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/0e3fc45c", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/1930s-d12-r11.25-4096ctx-run1", - "dataset_fingerprint": "ea9ba957b6e61d1c", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "unique_train_tokens_per_source": { - "original": 892647506, - "midtrain_r30": 255427871, - "midtrain_r60": 127443928 - }, - "unique_train_tokens": 1275519305, - "effective_epochs": 0.9708737854030363 -} diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer/experiment_tokenizer.json b/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer/token_bytes.pt b/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer/tokenizer.pkl b/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r11.25-4096ctx-run1/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_000500.json b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_000500.json deleted file mode 100644 index 534cd2a5471879ef46bd560d6c7802a56424c9b1..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,224 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "val_bpb": 1.3401288919896914, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "1930s-d12-r12-4096ctx-run2", - "wandb_run_id": "f20b8b2f", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio12,ctx4096,midtrain-schedule,fp8,seed43,variance-replicate", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "experiment_config": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c8a82511b832ffd2aa740a0b0453a8a118870493", - "seed": 43, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "1930s-d12-r12-4096ctx-run2", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r12-4096ctx-run2", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed43", - "variance-replicate" - ] - }, - "config_fingerprint": "8d2026ebce2c42d6", - "artifact_path": "experiments/1930s-d12-r12-4096ctx-run2" - }, - "stage": "base", - "base_experiment_id": "1930s-d12-r12-4096ctx-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "8d2026ebce2c42d6", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769, - "mixture": { - "cursors": { - "original": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 262176768, - "source_tokens": { - "original": 262176768, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 1.3401288919896914, - "smooth_train_loss": 3.693075330229297, - "total_training_time": 1487.517940044403, - "stage_training_flops": 291921016651776000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 291921016651776000 - } -} \ No newline at end of file diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_001000.json b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_001000.json deleted file mode 100644 index 7b6063e48c906fc63333a3f72f455e6175d1769f..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,224 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "val_bpb": 1.233793940888148, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "1930s-d12-r12-4096ctx-run2", - "wandb_run_id": "f20b8b2f", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio12,ctx4096,midtrain-schedule,fp8,seed43,variance-replicate", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "experiment_config": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c8a82511b832ffd2aa740a0b0453a8a118870493", - "seed": 43, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "1930s-d12-r12-4096ctx-run2", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r12-4096ctx-run2", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed43", - "variance-replicate" - ] - }, - "config_fingerprint": "8d2026ebce2c42d6", - "artifact_path": "experiments/1930s-d12-r12-4096ctx-run2" - }, - "stage": "base", - "base_experiment_id": "1930s-d12-r12-4096ctx-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "8d2026ebce2c42d6", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769, - "mixture": { - "cursors": { - "original": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 524320768, - "source_tokens": { - "original": 524320768, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 1.233793940888148, - "smooth_train_loss": 3.516132337256357, - "total_training_time": 3005.8546483516693, - "stage_training_flops": 583842033303552000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 583842033303552000 - } -} \ No newline at end of file diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_001500.json b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_001500.json deleted file mode 100644 index fccb527f30d6a0480951202977d8786f490acaad..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,224 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "val_bpb": 1.151211077254197, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "1930s-d12-r12-4096ctx-run2", - "wandb_run_id": "f20b8b2f", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio12,ctx4096,midtrain-schedule,fp8,seed43,variance-replicate", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "experiment_config": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c8a82511b832ffd2aa740a0b0453a8a118870493", - "seed": 43, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "1930s-d12-r12-4096ctx-run2", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r12-4096ctx-run2", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed43", - "variance-replicate" - ] - }, - "config_fingerprint": "8d2026ebce2c42d6", - "artifact_path": "experiments/1930s-d12-r12-4096ctx-run2" - }, - "stage": "base", - "base_experiment_id": "1930s-d12-r12-4096ctx-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "8d2026ebce2c42d6", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769, - "mixture": { - "cursors": { - "original": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 786464768, - "source_tokens": { - "original": 786464768, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 1.151211077254197, - "smooth_train_loss": 3.4062644418303023, - "total_training_time": 4525.468638658524, - "stage_training_flops": 875763049955328000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 875763049955328000 - } -} \ No newline at end of file diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_002000.json b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_002000.json deleted file mode 100644 index 27639a4f29eddf10f2bd419608bfd1d233cdb962..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,224 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "val_bpb": 1.1244294047643, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "1930s-d12-r12-4096ctx-run2", - "wandb_run_id": "f20b8b2f", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio12,ctx4096,midtrain-schedule,fp8,seed43,variance-replicate", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "experiment_config": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c8a82511b832ffd2aa740a0b0453a8a118870493", - "seed": 43, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "1930s-d12-r12-4096ctx-run2", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r12-4096ctx-run2", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed43", - "variance-replicate" - ] - }, - "config_fingerprint": "8d2026ebce2c42d6", - "artifact_path": "experiments/1930s-d12-r12-4096ctx-run2" - }, - "stage": "base", - "base_experiment_id": "1930s-d12-r12-4096ctx-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "8d2026ebce2c42d6", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 1, - "pos": 23768513, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 23768513, - "mixture": { - "cursors": { - "original": { - "file_idx": 9, - "pos": 24872256, - "epoch": 1, - "pq_idx": 9, - "rg_idx": 24872256 - }, - "midtrain_r30": { - "file_idx": 1, - "pos": 23768513, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 23768513 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 1048608768, - "source_tokens": { - "original": 924844032, - "midtrain_r30": 123764736, - "midtrain_r60": 0 - }, - "active_stage_idx": 1 - } - }, - "loop_state": { - "min_val_bpb": 1.1244294047643, - "smooth_train_loss": 3.1648974658165985, - "total_training_time": 6046.092838048935, - "stage_training_flops": 1167684066607104000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1167684066607104000 - } -} \ No newline at end of file diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_002500.json b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_002500.json deleted file mode 100644 index 3f62d4e8c5333916140c3e9945bd854e480a4252..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_002500.json +++ /dev/null @@ -1,224 +0,0 @@ -{ - "step": 2500, - "training_complete": false, - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "val_bpb": 1.0887535657197098, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "1930s-d12-r12-4096ctx-run2", - "wandb_run_id": "f20b8b2f", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio12,ctx4096,midtrain-schedule,fp8,seed43,variance-replicate", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "experiment_config": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c8a82511b832ffd2aa740a0b0453a8a118870493", - "seed": 43, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "1930s-d12-r12-4096ctx-run2", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r12-4096ctx-run2", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed43", - "variance-replicate" - ] - }, - "config_fingerprint": "8d2026ebce2c42d6", - "artifact_path": "experiments/1930s-d12-r12-4096ctx-run2" - }, - "stage": "base", - "base_experiment_id": "1930s-d12-r12-4096ctx-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "8d2026ebce2c42d6", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 1, - "pos": 21671297, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 21671297, - "mixture": { - "cursors": { - "original": { - "file_idx": 9, - "pos": 24872256, - "epoch": 1, - "pq_idx": 9, - "rg_idx": 24872256 - }, - "midtrain_r30": { - "file_idx": 2, - "pos": 64249216, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 64249216 - }, - "midtrain_r60": { - "file_idx": 1, - "pos": 21671297, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 21671297 - } - }, - "cumulative_tokens": 1310752768, - "source_tokens": { - "original": 924844032, - "midtrain_r30": 264241152, - "midtrain_r60": 121667584 - }, - "active_stage_idx": 2 - } - }, - "loop_state": { - "min_val_bpb": 1.0887535657197098, - "smooth_train_loss": 3.091377881648148, - "total_training_time": 7565.57666015625, - "stage_training_flops": 1459605083258880000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1459605083258880000 - } -} \ No newline at end of file diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_002520.json b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_002520.json deleted file mode 100644 index c5efd764c21d6820ce7bbc2ce2467a08685967dd..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/meta_002520.json +++ /dev/null @@ -1,224 +0,0 @@ -{ - "step": 2520, - "training_complete": true, - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "val_bpb": 1.0879491465161388, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "1930s-d12-r12-4096ctx-run2", - "wandb_run_id": "f20b8b2f", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio12,ctx4096,midtrain-schedule,fp8,seed43,variance-replicate", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "experiment_config": "/content/nanochat_cache/experiments/1930s-d12-r12-4096ctx-run2/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c8a82511b832ffd2aa740a0b0453a8a118870493", - "seed": 43, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "1930s-d12-r12-4096ctx-run2", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r12-4096ctx-run2", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed43", - "variance-replicate" - ] - }, - "config_fingerprint": "8d2026ebce2c42d6", - "artifact_path": "experiments/1930s-d12-r12-4096ctx-run2" - }, - "stage": "base", - "base_experiment_id": "1930s-d12-r12-4096ctx-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "8d2026ebce2c42d6", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 1, - "pos": 32157377, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 32157377, - "mixture": { - "cursors": { - "original": { - "file_idx": 9, - "pos": 24872256, - "epoch": 1, - "pq_idx": 9, - "rg_idx": 24872256 - }, - "midtrain_r30": { - "file_idx": 2, - "pos": 64249216, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 64249216 - }, - "midtrain_r60": { - "file_idx": 1, - "pos": 32157377, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 32157377 - } - }, - "cumulative_tokens": 1321238528, - "source_tokens": { - "original": 924844032, - "midtrain_r30": 264241152, - "midtrain_r60": 132153344 - }, - "active_stage_idx": 2 - } - }, - "loop_state": { - "min_val_bpb": 1.0879491465161388, - "smooth_train_loss": 3.0810559880062787, - "total_training_time": 7626.332884550095, - "stage_training_flops": 1471281923924951040, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1471281923924951040 - } -} \ No newline at end of file diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_000500.pt b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_000500.pt deleted file mode 100644 index 31401e0fb57c7b82ace310af331d562c52f59e97..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0a52560dd47d1695371d4d712f7802b0f471cff42783d9308800cddbc803107f -size 792761690 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_001000.pt b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_001000.pt deleted file mode 100644 index d58a8cde574d67251ce85cb0584dc18d1895dce2..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a9b5396fb98b64dfeb7453f649fb50af2db8b07321adcb31799b4ce0d07fb8df -size 792761690 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_001500.pt b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_001500.pt deleted file mode 100644 index 79b2b9e24e4e21c52e6a6644974476cb7de5dcdb..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e649f008eaa71746b80257b66ee5ea665119f5c7b47f1ac2f420d23f2db75e52 -size 792761690 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_002000.pt b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_002000.pt deleted file mode 100644 index 71bf48bdcee8dd938f3d3d6dc0685df8b628be3e..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4758f65670f42c5d3379c588b2c64dba9527b431cd28285f8fad0abed5decbf9 -size 792761690 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_002500.pt b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_002500.pt deleted file mode 100644 index 3d0c554bacd92ad97c6f529e1b180b0ee8b57822..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fee8537f641d8ea2a2d2597fdb65657ebae335a860cd5555a10237412ae06c31 -size 792761690 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_002520.pt b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_002520.pt deleted file mode 100644 index f56f9246f49398cd33ddb18fb8e2daf493719640..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/model_002520.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:75942e18fe10abfeac0602fc135cbfbda91e9b73d64044e0dc3de9bab918c918 -size 792761690 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_000500_rank0.pt b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 81d7d3df2bf14e806e4239eb1edf6e6e0cf3f053..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4c16c8d84970f4ad6704cc0da4fb15bf47f9e20dddcdd8a25208bc4841b45d79 -size 1246165357 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_001000_rank0.pt b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index da1e7356f380ea9f23ef41eb16f2edd58712ff2c..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:30dc08a39b60544f59bb385cc933c57a96ce4136dcc9f8a5a7e3c1f10daca07c -size 1246165357 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_001500_rank0.pt b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 87428093a2e055748141bde81471776728335d0d..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a7a164eb49eba8e42de5bbe8e7bc580dcd0e67d1e8b5d79f0711278673a1d785 -size 1246165357 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_002000_rank0.pt b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index cd185441c94b0a2bd1bd1f9751ca7624e5fa86db..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:815688c45d1b1de062f86b18bf26dbcbd22196b82d1a4a2a13dfda66e1a8fe01 -size 1246165357 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_002500_rank0.pt b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_002500_rank0.pt deleted file mode 100644 index f4842b55193ce3888d7e6645b435c19a7fe00f4d..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:dd5effd9476ddcc4cb2fd3c8968e5ea9c3b18911874bc075ea04fcd094b4c31e -size 1246165357 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_002520_rank0.pt b/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_002520_rank0.pt deleted file mode 100644 index 9f6e3f2f7e25c5e5fa9206f3e4bbf6be7581ff1c..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/base_checkpoints/optim_002520_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:90621803f6152e5d946bd259a5eb8eea080493b10c52af2dc184ea53a2d5fc74 -size 1246165357 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/config.json b/experiments/1930s-d12-r12-4096ctx-run2/config.json deleted file mode 100644 index 37b6d93de6e3cdc580570081e7636cef7e8cc4b5..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/config.json +++ /dev/null @@ -1,107 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "1930s-d12-r12-4096ctx-run2", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8", - "seed43", - "variance-replicate" - ] - }, - "config_fingerprint": "8d2026ebce2c42d6", - "artifact_path": "experiments/1930s-d12-r12-4096ctx-run2" -} diff --git a/experiments/1930s-d12-r12-4096ctx-run2/evals/core.json b/experiments/1930s-d12-r12-4096ctx-run2/evals/core.json deleted file mode 100644 index cdd283fc3b972545ae550722a6b3fa4a2711e9b8..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2520)", - "step": 2520, - "bpb": {}, - "core_metric": 0.07523102469394939, - "core_results": { - "hellaswag_zeroshot": 0.2814180254936218, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.08995620161294937, - "arc_easy": 0.3291245698928833, - "arc_challenge": 0.2226962447166443, - "copa": 0.5399999618530273, - "commonsense_qa": 0.2981162965297699, - "piqa": 0.5609357953071594, - "openbook_qa": 0.23200000822544098, - "lambada_openai": 0.23578497767448425, - "hellaswag": 0.28261300921440125, - "winograd": 0.5567765831947327, - "winogrande": 0.5003946423530579, - "bigbench_dyck_languages": 0.11300000548362732, - "agi_eval_lsat_ar": 0.2913043200969696, - "bigbench_cs_algorithms": 0.3643939197063446, - "bigbench_operators": 0.06666667014360428, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.013528854586184025, - "coqa": 0.0751597136259079, - "boolq": 0.5581039786338806, - "bigbench_language_identification": 0.25099998712539673 - }, - "centered_results": { - "hellaswag_zeroshot": 0.04189070065816244, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.08995620161294937, - "arc_easy": 0.1054994265238444, - "arc_challenge": -0.03640500704447428, - "copa": 0.07999992370605469, - "commonsense_qa": 0.12264537066221236, - "piqa": 0.12187159061431885, - "openbook_qa": -0.02399998903274536, - "lambada_openai": 0.23578497767448425, - "hellaswag": 0.043484012285868325, - "winograd": 0.11355316638946533, - "winogrande": 0.0007892847061157227, - "bigbench_dyck_languages": 0.11300000548362732, - "agi_eval_lsat_ar": 0.11413040012121199, - "bigbench_cs_algorithms": 0.3643939197063446, - "bigbench_operators": 0.06666667014360428, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.013528854586184025, - "coqa": 0.0751597136259079, - "boolq": -0.16288426675294573, - "bigbench_language_identification": 0.17601758759669606 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/1930s-d12-r12-4096ctx-run2/evals/samples.json b/experiments/1930s-d12-r12-4096ctx-run2/evals/samples.json deleted file mode 100644 index 19c63993360f6efb948471e7920d6a520c6802ab..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2520)", - "step": 2520, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the United States. The capital of the United States is the capital" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the gold of the earth. The gold of the earth is the" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. The day is the same as the day of the week, and the" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is a hot, and the opposite of cold is a cold. The former is a" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the sun, the moon, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a dark brown, with a dark spot on the back of the head. The" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the" - } - ], - "unconditioned_samples": [ - "<|bos|>J. MacDonald.\n\nStatistics .. 4.50\n\nNeubulus Plantarum 3.00\n\nVarieties '^3 X 25.00\n\nKT 2\n\nAmericana Leucos 1.50 COUNTA FLORALULATA.\n\nType of man in size, other similar Structural phenomena, Diagram 1 1 1 0 . 5...32 .10 . 20 . 44 10.30 . 60 20 \u00b028 . 35 10 /25 27 10 .00 . 30 40 50 .02", - "<|bos|>Ed. London, 1812, p. 357. 21\n\n1263. Climate and Influenza. - This widespread epidemic disease has proved to be one of the most fatal of all the human underT6 _ diseases of man. According to Kodzkin, measles, scarlet fever, smallpox, and other contagious diseases and diseases of the person gradually become fatal. Persons dealing with persons suspected of measles or scarlet fever, or those suffering from various infectious diseases, whom reviewers have heard called \" idiotic families,\" are included in this great category. But there are also those who", - "<|bos|>Charm's Tenements is large and rapid; The Cranes also move rapidly; The Diogenes moves quicker than the\n\nBox Tenements in Favor of Sponge.\"\n\nModernist odeists rapidly equip piants greatly in numbers,\"\n\nAmong the earliest in cison. At the present time they are in considerable numbers. Henle's Tenements (?) in 1891 appeared at a very high rate of advance. Storm increases the number of coaches annually.\n\nAt Blankham, Sauqs, Mouqs, and the French posts are large and ready to get in. Cesse is only about", - "<|bos|>Hires 350,752\n\nION.\n\n12 CH\n\nJUDGMENT\n\n9 I\n\nJUDGMENT\n\n11 I A\n\nCOLUMNS\n\nCALL OF HALF-HOW THAT HORN OF BEST PLEADNIKLI BONDS, HOLD SICKNESS, EPULTHESTA DAY, MACYDHIA 50 Enter DISPRISEPS, STONEOS, CONCORD HOME, SKETONING\n\nIt is the fair and pleasant Hibernia, Like sister well-pressed and pretty, that her pictures adorn;\n\nThus with her needle glean Sunday lies;\n\nOn ours that", - "<|bos|>eed\n\nCaswell paid them for the wheat, and seft i post gave a notice of their intention to pro- Scork had tenanted\n\nCrown Island, and lb-No, find it was for to imprison before taking affidavits, and offing great odds; and absent' private in Drj: Watts' time he had to come of post when in the spring before estate on Douglass lunged; have thought no time remained fifteen Months without complaining as to deposition on oath or affirmation;th Idum: could not find the housevantages the estate on Champlain that has the most of Ahembie the Sir=for", - "<|bos|>PREFACE\n\nand require. This. I can hardly accomplish. being able to fulfil my expressed purpose, and that of frankly acknowledging my indebtedness at least, but of asking for information of the furthest value, and at first glance. It is only in one place that most of this provision is inserted.\n\nA political architect might tire of the details of a Fish Commission much longer and often longer in Ireland, and could criticise expressions of reform from every-day affairs. We find, for example, in one of the National Oriental Libraries an interesting official examination of a Block building in all respects complete.\n\nIs there, in the Professor's evidence,", - "<|bos|>Clarke. He repeated the attempt to wrest Hitmarsh to be thrown into the proper place; whereupon there were subpoenas, enjoining such necessary preparations as a rushing wind could suggest. The latter was opposed; for a smell of burning sulphur in it did simply add to the trouble. Then, after wearing so many ropes about him, they were tied with ropes and drags of blue cloth and led away to Woburn, where it would have beene reserved for the mob to have turned their weapons into hempen strings of beads, and substituted clasp-knives and ropes like to pieces of wood for cloaks out of Plymouth", - "<|bos|>Hist. Ver.\n\n1 The rerew of a large domain, that gases absorbed and the tone of the plant in the use of the mineral matter overcoming large quantities\n\nof mineral matter are easily converted into the earth.\n\nIt is the man who constructs or makes a portion of the great mountain, and so fulfills his purpose. If, however, we look closer, and see how slowly and imperfectly the law of matter is acted upon by elective or experimentally easily effused forces in nature, then we can scarcely say the same things as when Nature is comprehended under her narrow but tremendous definition of beauty. It is an ideal revelation. In" - ] -} \ No newline at end of file diff --git a/experiments/1930s-d12-r12-4096ctx-run2/evals/val_bpb.json b/experiments/1930s-d12-r12-4096ctx-run2/evals/val_bpb.json deleted file mode 100644 index f3b1a4c3955930f26130a8878ed0f3eac981701f..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2520)", - "step": 2520, - "bpb": { - "val": 1.0332056972909351 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/1930s-d12-r12-4096ctx-run2/run.json b/experiments/1930s-d12-r12-4096ctx-run2/run.json deleted file mode 100644 index ce3a731ea536a68cb255aca41e218b2c13c6f7c8..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "stage": "base", - "base_experiment_id": "1930s-d12-r12-4096ctx-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "8d2026ebce2c42d6", - "wandb_run_id": "f20b8b2f", - "created_at": 1785170620 -} diff --git a/experiments/1930s-d12-r12-4096ctx-run2/summary.json b/experiments/1930s-d12-r12-4096ctx-run2/summary.json deleted file mode 100644 index e15c43da483a73c027d497f542c6d56e9fe49d88..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/summary.json +++ /dev/null @@ -1,98 +0,0 @@ -{ - "experiment_id": "1930s-d12-r12-4096ctx-run2", - "stage": "base", - "base_experiment_id": "1930s-d12-r12-4096ctx-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean-1930s", - "dataset_revision": "main", - "step": 2520, - "depth": 12, - "target_param_data_ratio": 12, - "training_tokens": 1321205760, - "final_sampled_val_bpb": 1.0879491465161388, - "minimum_sampled_val_bpb": 1.0879491465161388, - "full_val_bpb": 1.0332056972909351, - "core_metric": 0.07523102469394939, - "centered_results": { - "hellaswag_zeroshot": 0.04189070065816244, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.08995620161294937, - "arc_easy": 0.1054994265238444, - "arc_challenge": -0.03640500704447428, - "copa": 0.07999992370605469, - "commonsense_qa": 0.12264537066221236, - "piqa": 0.12187159061431885, - "openbook_qa": -0.02399998903274536, - "lambada_openai": 0.23578497767448425, - "hellaswag": 0.043484012285868325, - "winograd": 0.11355316638946533, - "winogrande": 0.0007892847061157227, - "bigbench_dyck_languages": 0.11300000548362732, - "agi_eval_lsat_ar": 0.11413040012121199, - "bigbench_cs_algorithms": 0.3643939197063446, - "bigbench_operators": 0.06666667014360428, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.013528854586184025, - "coqa": 0.0751597136259079, - "boolq": -0.16288426675294573, - "bigbench_language_identification": 0.17601758759669606 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the United States. The capital of the United States is the capital" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the gold of the earth. The gold of the earth is the" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. The day is the same as the day of the week, and the" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is a hot, and the opposite of cold is a cold. The former is a" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the sun, the moon, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a dark brown, with a dark spot on the back of the head. The" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the" - } - ], - "unconditioned_samples": [ - "<|bos|>J. MacDonald.\n\nStatistics .. 4.50\n\nNeubulus Plantarum 3.00\n\nVarieties '^3 X 25.00\n\nKT 2\n\nAmericana Leucos 1.50 COUNTA FLORALULATA.\n\nType of man in size, other similar Structural phenomena, Diagram 1 1 1 0 . 5...32 .10 . 20 . 44 10.30 . 60 20 \u00b028 . 35 10 /25 27 10 .00 . 30 40 50 .02", - "<|bos|>Ed. London, 1812, p. 357. 21\n\n1263. Climate and Influenza. - This widespread epidemic disease has proved to be one of the most fatal of all the human underT6 _ diseases of man. According to Kodzkin, measles, scarlet fever, smallpox, and other contagious diseases and diseases of the person gradually become fatal. Persons dealing with persons suspected of measles or scarlet fever, or those suffering from various infectious diseases, whom reviewers have heard called \" idiotic families,\" are included in this great category. But there are also those who", - "<|bos|>Charm's Tenements is large and rapid; The Cranes also move rapidly; The Diogenes moves quicker than the\n\nBox Tenements in Favor of Sponge.\"\n\nModernist odeists rapidly equip piants greatly in numbers,\"\n\nAmong the earliest in cison. At the present time they are in considerable numbers. Henle's Tenements (?) in 1891 appeared at a very high rate of advance. Storm increases the number of coaches annually.\n\nAt Blankham, Sauqs, Mouqs, and the French posts are large and ready to get in. Cesse is only about", - "<|bos|>Hires 350,752\n\nION.\n\n12 CH\n\nJUDGMENT\n\n9 I\n\nJUDGMENT\n\n11 I A\n\nCOLUMNS\n\nCALL OF HALF-HOW THAT HORN OF BEST PLEADNIKLI BONDS, HOLD SICKNESS, EPULTHESTA DAY, MACYDHIA 50 Enter DISPRISEPS, STONEOS, CONCORD HOME, SKETONING\n\nIt is the fair and pleasant Hibernia, Like sister well-pressed and pretty, that her pictures adorn;\n\nThus with her needle glean Sunday lies;\n\nOn ours that", - "<|bos|>eed\n\nCaswell paid them for the wheat, and seft i post gave a notice of their intention to pro- Scork had tenanted\n\nCrown Island, and lb-No, find it was for to imprison before taking affidavits, and offing great odds; and absent' private in Drj: Watts' time he had to come of post when in the spring before estate on Douglass lunged; have thought no time remained fifteen Months without complaining as to deposition on oath or affirmation;th Idum: could not find the housevantages the estate on Champlain that has the most of Ahembie the Sir=for", - "<|bos|>PREFACE\n\nand require. This. I can hardly accomplish. being able to fulfil my expressed purpose, and that of frankly acknowledging my indebtedness at least, but of asking for information of the furthest value, and at first glance. It is only in one place that most of this provision is inserted.\n\nA political architect might tire of the details of a Fish Commission much longer and often longer in Ireland, and could criticise expressions of reform from every-day affairs. We find, for example, in one of the National Oriental Libraries an interesting official examination of a Block building in all respects complete.\n\nIs there, in the Professor's evidence,", - "<|bos|>Clarke. He repeated the attempt to wrest Hitmarsh to be thrown into the proper place; whereupon there were subpoenas, enjoining such necessary preparations as a rushing wind could suggest. The latter was opposed; for a smell of burning sulphur in it did simply add to the trouble. Then, after wearing so many ropes about him, they were tied with ropes and drags of blue cloth and led away to Woburn, where it would have beene reserved for the mob to have turned their weapons into hempen strings of beads, and substituted clasp-knives and ropes like to pieces of wood for cloaks out of Plymouth", - "<|bos|>Hist. Ver.\n\n1 The rerew of a large domain, that gases absorbed and the tone of the plant in the use of the mineral matter overcoming large quantities\n\nof mineral matter are easily converted into the earth.\n\nIt is the man who constructs or makes a portion of the great mountain, and so fulfills his purpose. If, however, we look closer, and see how slowly and imperfectly the law of matter is acted upon by elective or experimentally easily effused forces in nature, then we can scarcely say the same things as when Nature is comprehended under her narrow but tremendous definition of beauty. It is an ideal revelation. In" - ], - "training_time_seconds": 7626.332884550095, - "stage_training_flops": 1.471281923924951e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.471281923924951e+18, - "config_fingerprint": "8d2026ebce2c42d6", - "git_commit_sha": "c8a82511b832ffd2aa740a0b0453a8a118870493", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/f20b8b2f", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/1930s-d12-r12-4096ctx-run2", - "dataset_fingerprint": "ea9ba957b6e61d1c", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "unique_train_tokens_per_source": { - "original": 952589353, - "midtrain_r30": 272168387, - "midtrain_r60": 136084194 - }, - "unique_train_tokens": 1360841934, - "effective_epochs": 0.9708737855516436 -} diff --git a/experiments/1930s-d12-r12-4096ctx-run2/tokenizer/experiment_tokenizer.json b/experiments/1930s-d12-r12-4096ctx-run2/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/1930s-d12-r12-4096ctx-run2/tokenizer/token_bytes.pt b/experiments/1930s-d12-r12-4096ctx-run2/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/1930s-d12-r12-4096ctx-run2/tokenizer/tokenizer.pkl b/experiments/1930s-d12-r12-4096ctx-run2/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/1930s-d12-r12-4096ctx-run2/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_006000.json b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_006000.json deleted file mode 100644 index 48618142c7276b3d6baa880774f64c2a65941b0c..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_006000.json +++ /dev/null @@ -1,244 +0,0 @@ -{ - "step": 6000, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "val_bpb": 0.8878000321507976, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont", - "wandb_run_id": "2fa1b744", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-v2,0-21-45,midtrain-2.88ep,data-repair-continuation", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "init_from_step": 5500, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "Think.Unbounded-d32", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_original\", \"midtrain_r21\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r21\", \"midtrain_r45\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r45\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "40dd55cc291e41cf70bdb8b3df90f4bc32665d8d", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32-v2mix-cont", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "branch": { - "parent_experiment_id": "Think.Unbounded-d32", - "parent_step": 5500, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r21": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_21/data", - "validation_shard": 65, - "num_train_shards": 65, - "download_workers": 4 - }, - "midtrain_r45": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_45/data", - "validation_shard": 33, - "num_train_shards": 33, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r21" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r45" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.02, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32-v2mix-cont", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-v2", - "0-21-45", - "midtrain-2.88ep", - "data-repair-continuation" - ] - }, - "config_fingerprint": "00af0bde578c1269", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "config_fingerprint": "00af0bde578c1269" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 125, - "pos": 84464386, - "epoch": 1, - "pq_idx": 125, - "rg_idx": 84464386, - "mixture": { - "cursors": { - "original": { - "file_idx": 125, - "pos": 84464386, - "epoch": 1, - "pq_idx": 125, - "rg_idx": 84464386 - }, - "midtrain_r21": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r45": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 12582928384, - "source_tokens": { - "original": 12582928384, - "midtrain_r21": 0, - "midtrain_r45": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 0.8878000321507976, - "smooth_train_loss": 2.269974019977436, - "total_training_time": 19584.64479827881, - "stage_start_step": 5500, - "stage_training_flops": 12033074581733376000, - "inherited_parent_flops": 1.3236382039906714e+20, - "cumulative_pipeline_training_flops": 1.4439689498080051e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_006500.json b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_006500.json deleted file mode 100644 index a7d6a9b08def04bdda0add7e05556950a59ad82a..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_006500.json +++ /dev/null @@ -1,244 +0,0 @@ -{ - "step": 6500, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "val_bpb": 0.8884785698076477, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont", - "wandb_run_id": "2fa1b744", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-v2,0-21-45,midtrain-2.88ep,data-repair-continuation", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "init_from_step": 5500, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "Think.Unbounded-d32", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_original\", \"midtrain_r21\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r21\", \"midtrain_r45\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r45\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "40dd55cc291e41cf70bdb8b3df90f4bc32665d8d", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32-v2mix-cont", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "branch": { - "parent_experiment_id": "Think.Unbounded-d32", - "parent_step": 5500, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r21": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_21/data", - "validation_shard": 65, - "num_train_shards": 65, - "download_workers": 4 - }, - "midtrain_r45": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_45/data", - "validation_shard": 33, - "num_train_shards": 33, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r21" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r45" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.02, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32-v2mix-cont", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-v2", - "0-21-45", - "midtrain-2.88ep", - "data-repair-continuation" - ] - }, - "config_fingerprint": "00af0bde578c1269", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "config_fingerprint": "00af0bde578c1269" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 136, - "pos": 33168386, - "epoch": 1, - "pq_idx": 136, - "rg_idx": 33168386, - "mixture": { - "cursors": { - "original": { - "file_idx": 136, - "pos": 33168386, - "epoch": 1, - "pq_idx": 136, - "rg_idx": 33168386 - }, - "midtrain_r21": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r45": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 13631504384, - "source_tokens": { - "original": 13631504384, - "midtrain_r21": 0, - "midtrain_r45": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 0.8878000321507976, - "smooth_train_loss": 2.492017514832503, - "total_training_time": 39617.187046051025, - "stage_start_step": 5500, - "stage_training_flops": 24066149163466752000, - "inherited_parent_flops": 1.3236382039906714e+20, - "cumulative_pipeline_training_flops": 1.564299695625339e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_007000.json b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_007000.json deleted file mode 100644 index d590eb95c84b5032a953d3548487eca542dad210..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_007000.json +++ /dev/null @@ -1,244 +0,0 @@ -{ - "step": 7000, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "val_bpb": 0.8783667228413535, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont", - "wandb_run_id": "2fa1b744", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-v2,0-21-45,midtrain-2.88ep,data-repair-continuation", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "init_from_step": 5500, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "Think.Unbounded-d32", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_original\", \"midtrain_r21\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r21\", \"midtrain_r45\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r45\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "40dd55cc291e41cf70bdb8b3df90f4bc32665d8d", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32-v2mix-cont", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "branch": { - "parent_experiment_id": "Think.Unbounded-d32", - "parent_step": 5500, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r21": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_21/data", - "validation_shard": 65, - "num_train_shards": 65, - "download_workers": 4 - }, - "midtrain_r45": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_45/data", - "validation_shard": 33, - "num_train_shards": 33, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r21" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r45" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.02, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32-v2mix-cont", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-v2", - "0-21-45", - "midtrain-2.88ep", - "data-repair-continuation" - ] - }, - "config_fingerprint": "00af0bde578c1269", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "config_fingerprint": "00af0bde578c1269" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 87290626, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 87290626, - "mixture": { - "cursors": { - "original": { - "file_idx": 140, - "pos": 94581760, - "epoch": 1, - "pq_idx": 140, - "rg_idx": 94581760 - }, - "midtrain_r21": { - "file_idx": 5, - "pos": 87290626, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 87290626 - }, - "midtrain_r45": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 14680080384, - "source_tokens": { - "original": 14092861440, - "midtrain_r21": 587218944, - "midtrain_r45": 0 - }, - "active_stage_idx": 1 - } - }, - "loop_state": { - "min_val_bpb": 0.8783667228413535, - "smooth_train_loss": 2.5095336084649387, - "total_training_time": 59663.125319480896, - "stage_start_step": 5500, - "stage_training_flops": 36099223745200128000, - "inherited_parent_flops": 1.3236382039906714e+20, - "cumulative_pipeline_training_flops": 1.6846304414426726e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_007500.json b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_007500.json deleted file mode 100644 index 71ac3fcacabea036a60776ede89125f8d4c158df..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_007500.json +++ /dev/null @@ -1,244 +0,0 @@ -{ - "step": 7500, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "val_bpb": 0.8710382187147858, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont", - "wandb_run_id": "2fa1b744", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-v2,0-21-45,midtrain-2.88ep,data-repair-continuation", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "init_from_step": 5500, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "Think.Unbounded-d32", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_original\", \"midtrain_r21\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r21\", \"midtrain_r45\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r45\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "40dd55cc291e41cf70bdb8b3df90f4bc32665d8d", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32-v2mix-cont", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "branch": { - "parent_experiment_id": "Think.Unbounded-d32", - "parent_step": 5500, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r21": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_21/data", - "validation_shard": 65, - "num_train_shards": 65, - "download_workers": 4 - }, - "midtrain_r45": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_45/data", - "validation_shard": 33, - "num_train_shards": 33, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r21" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r45" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.02, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32-v2mix-cont", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-v2", - "0-21-45", - "midtrain-2.88ep", - "data-repair-continuation" - ] - }, - "config_fingerprint": "00af0bde578c1269", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "config_fingerprint": "00af0bde578c1269" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 16, - "pos": 35994626, - "epoch": 1, - "pq_idx": 16, - "rg_idx": 35994626, - "mixture": { - "cursors": { - "original": { - "file_idx": 140, - "pos": 94581760, - "epoch": 1, - "pq_idx": 140, - "rg_idx": 94581760 - }, - "midtrain_r21": { - "file_idx": 16, - "pos": 35994626, - "epoch": 1, - "pq_idx": 16, - "rg_idx": 35994626 - }, - "midtrain_r45": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 15728656384, - "source_tokens": { - "original": 14092861440, - "midtrain_r21": 1635794944, - "midtrain_r45": 0 - }, - "active_stage_idx": 1 - } - }, - "loop_state": { - "min_val_bpb": 0.8684623742403045, - "smooth_train_loss": 2.589196119655139, - "total_training_time": 79736.70196843147, - "stage_start_step": 5500, - "stage_training_flops": 48132298326933504000, - "inherited_parent_flops": 1.3236382039906714e+20, - "cumulative_pipeline_training_flops": 1.8049611872600064e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_008000.json b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_008000.json deleted file mode 100644 index 20b889de99516406ac031948ba8db7ae545c6430..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_008000.json +++ /dev/null @@ -1,244 +0,0 @@ -{ - "step": 8000, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "val_bpb": 0.8427847059243602, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont", - "wandb_run_id": "2fa1b744", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-v2,0-21-45,midtrain-2.88ep,data-repair-continuation", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "init_from_step": 5500, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "Think.Unbounded-d32", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_original\", \"midtrain_r21\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r21\", \"midtrain_r45\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r45\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "40dd55cc291e41cf70bdb8b3df90f4bc32665d8d", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32-v2mix-cont", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "branch": { - "parent_experiment_id": "Think.Unbounded-d32", - "parent_step": 5500, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r21": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_21/data", - "validation_shard": 65, - "num_train_shards": 65, - "download_workers": 4 - }, - "midtrain_r45": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_45/data", - "validation_shard": 33, - "num_train_shards": 33, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r21" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r45" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.02, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32-v2mix-cont", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-v2", - "0-21-45", - "midtrain-2.88ep", - "data-repair-continuation" - ] - }, - "config_fingerprint": "00af0bde578c1269", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "config_fingerprint": "00af0bde578c1269" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 26, - "pos": 84698626, - "epoch": 1, - "pq_idx": 26, - "rg_idx": 84698626, - "mixture": { - "cursors": { - "original": { - "file_idx": 140, - "pos": 94581760, - "epoch": 1, - "pq_idx": 140, - "rg_idx": 94581760 - }, - "midtrain_r21": { - "file_idx": 26, - "pos": 84698626, - "epoch": 1, - "pq_idx": 26, - "rg_idx": 84698626 - }, - "midtrain_r45": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 16777232384, - "source_tokens": { - "original": 14092861440, - "midtrain_r21": 2684370944, - "midtrain_r45": 0 - }, - "active_stage_idx": 1 - } - }, - "loop_state": { - "min_val_bpb": 0.8427847059243602, - "smooth_train_loss": 2.1792102215093845, - "total_training_time": 99832.61976385117, - "stage_start_step": 5500, - "stage_training_flops": 60165372908666880000, - "inherited_parent_flops": 1.3236382039906714e+20, - "cumulative_pipeline_training_flops": 1.92529193307734e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_008500.json b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_008500.json deleted file mode 100644 index 802654f834de5babbfcce06a43cade02890c36e1..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_008500.json +++ /dev/null @@ -1,244 +0,0 @@ -{ - "step": 8500, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "val_bpb": 0.8322083944920666, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont", - "wandb_run_id": "2fa1b744", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-v2,0-21-45,midtrain-2.88ep,data-repair-continuation", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "init_from_step": 5500, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "Think.Unbounded-d32", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_original\", \"midtrain_r21\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r21\", \"midtrain_r45\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r45\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "40dd55cc291e41cf70bdb8b3df90f4bc32665d8d", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32-v2mix-cont", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "branch": { - "parent_experiment_id": "Think.Unbounded-d32", - "parent_step": 5500, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r21": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_21/data", - "validation_shard": 65, - "num_train_shards": 65, - "download_workers": 4 - }, - "midtrain_r45": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_45/data", - "validation_shard": 33, - "num_train_shards": 33, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r21" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r45" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.02, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32-v2mix-cont", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-v2", - "0-21-45", - "midtrain-2.88ep", - "data-repair-continuation" - ] - }, - "config_fingerprint": "00af0bde578c1269", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "config_fingerprint": "00af0bde578c1269" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 37, - "pos": 33402626, - "epoch": 1, - "pq_idx": 37, - "rg_idx": 33402626, - "mixture": { - "cursors": { - "original": { - "file_idx": 140, - "pos": 94581760, - "epoch": 1, - "pq_idx": 140, - "rg_idx": 94581760 - }, - "midtrain_r21": { - "file_idx": 37, - "pos": 33402626, - "epoch": 1, - "pq_idx": 37, - "rg_idx": 33402626 - }, - "midtrain_r45": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 17825808384, - "source_tokens": { - "original": 14092861440, - "midtrain_r21": 3732946944, - "midtrain_r45": 0 - }, - "active_stage_idx": 1 - } - }, - "loop_state": { - "min_val_bpb": 0.8322083944920666, - "smooth_train_loss": 2.329752848437703, - "total_training_time": 119936.27257323265, - "stage_start_step": 5500, - "stage_training_flops": 72198447490400256000, - "inherited_parent_flops": 1.3236382039906714e+20, - "cumulative_pipeline_training_flops": 2.045622678894674e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_009000.json b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_009000.json deleted file mode 100644 index ecf0842f37f948ac79f39d36888ed935307b66ab..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_009000.json +++ /dev/null @@ -1,244 +0,0 @@ -{ - "step": 9000, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "val_bpb": 0.804359703106789, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont", - "wandb_run_id": "2fa1b744", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-v2,0-21-45,midtrain-2.88ep,data-repair-continuation", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "init_from_step": 5500, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "Think.Unbounded-d32", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_original\", \"midtrain_r21\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r21\", \"midtrain_r45\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r45\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "40dd55cc291e41cf70bdb8b3df90f4bc32665d8d", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32-v2mix-cont", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "branch": { - "parent_experiment_id": "Think.Unbounded-d32", - "parent_step": 5500, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r21": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_21/data", - "validation_shard": 65, - "num_train_shards": 65, - "download_workers": 4 - }, - "midtrain_r45": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_45/data", - "validation_shard": 33, - "num_train_shards": 33, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r21" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r45" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.02, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32-v2mix-cont", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-v2", - "0-21-45", - "midtrain-2.88ep", - "data-repair-continuation" - ] - }, - "config_fingerprint": "00af0bde578c1269", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "config_fingerprint": "00af0bde578c1269" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 55083266, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 55083266, - "mixture": { - "cursors": { - "original": { - "file_idx": 140, - "pos": 94581760, - "epoch": 1, - "pq_idx": 140, - "rg_idx": 94581760 - }, - "midtrain_r21": { - "file_idx": 40, - "pos": 27023360, - "epoch": 1, - "pq_idx": 40, - "rg_idx": 27023360 - }, - "midtrain_r45": { - "file_idx": 7, - "pos": 55083266, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 55083266 - } - }, - "cumulative_tokens": 18874384384, - "source_tokens": { - "original": 14092861440, - "midtrain_r21": 4026531840, - "midtrain_r45": 754991104 - }, - "active_stage_idx": 2 - } - }, - "loop_state": { - "min_val_bpb": 0.804359703106789, - "smooth_train_loss": 2.1602357994357506, - "total_training_time": 139986.46024250984, - "stage_start_step": 5500, - "stage_training_flops": 84231522072133632000, - "inherited_parent_flops": 1.3236382039906714e+20, - "cumulative_pipeline_training_flops": 2.1659534247120077e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_009500.json b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_009500.json deleted file mode 100644 index 2442db07b81ec93397e59fd4355f8c96f9f962df..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_009500.json +++ /dev/null @@ -1,244 +0,0 @@ -{ - "step": 9500, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "val_bpb": 0.7701763237939314, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont", - "wandb_run_id": "2fa1b744", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-v2,0-21-45,midtrain-2.88ep,data-repair-continuation", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "init_from_step": 5500, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "Think.Unbounded-d32", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_original\", \"midtrain_r21\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r21\", \"midtrain_r45\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r45\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "40dd55cc291e41cf70bdb8b3df90f4bc32665d8d", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32-v2mix-cont", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "branch": { - "parent_experiment_id": "Think.Unbounded-d32", - "parent_step": 5500, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r21": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_21/data", - "validation_shard": 65, - "num_train_shards": 65, - "download_workers": 4 - }, - "midtrain_r45": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_45/data", - "validation_shard": 33, - "num_train_shards": 33, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r21" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r45" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.02, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32-v2mix-cont", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-v2", - "0-21-45", - "midtrain-2.88ep", - "data-repair-continuation" - ] - }, - "config_fingerprint": "00af0bde578c1269", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "config_fingerprint": "00af0bde578c1269" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 18, - "pos": 3787266, - "epoch": 1, - "pq_idx": 18, - "rg_idx": 3787266, - "mixture": { - "cursors": { - "original": { - "file_idx": 140, - "pos": 94581760, - "epoch": 1, - "pq_idx": 140, - "rg_idx": 94581760 - }, - "midtrain_r21": { - "file_idx": 40, - "pos": 27023360, - "epoch": 1, - "pq_idx": 40, - "rg_idx": 27023360 - }, - "midtrain_r45": { - "file_idx": 18, - "pos": 3787266, - "epoch": 1, - "pq_idx": 18, - "rg_idx": 3787266 - } - }, - "cumulative_tokens": 19922960384, - "source_tokens": { - "original": 14092861440, - "midtrain_r21": 4026531840, - "midtrain_r45": 1803567104 - }, - "active_stage_idx": 2 - } - }, - "loop_state": { - "min_val_bpb": 0.7701763237939314, - "smooth_train_loss": 2.2726057068037204, - "total_training_time": 160031.62036943436, - "stage_start_step": 5500, - "stage_training_flops": 96264596653867008000, - "inherited_parent_flops": 1.3236382039906714e+20, - "cumulative_pipeline_training_flops": 2.2862841705293414e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_009600.json b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_009600.json deleted file mode 100644 index 13af20258d72fb024036df6fc3ffedc2c8691a28..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/meta_009600.json +++ /dev/null @@ -1,244 +0,0 @@ -{ - "step": 9600, - "training_complete": true, - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "val_bpb": 0.7708812390249374, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont", - "wandb_run_id": "2fa1b744", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-v2,0-21-45,midtrain-2.88ep,data-repair-continuation", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "init_from_step": 5500, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "Think.Unbounded-d32", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_original\", \"midtrain_r21\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r21\", \"midtrain_r45\": \"/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/pretok_midtrain_r45\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "40dd55cc291e41cf70bdb8b3df90f4bc32665d8d", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32-v2mix-cont", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "branch": { - "parent_experiment_id": "Think.Unbounded-d32", - "parent_step": 5500, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r21": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_21/data", - "validation_shard": 65, - "num_train_shards": 65, - "download_workers": 4 - }, - "midtrain_r45": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_45/data", - "validation_shard": 33, - "num_train_shards": 33, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r21" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r45" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.02, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32-v2mix-cont", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-v2", - "0-21-45", - "midtrain-2.88ep", - "data-repair-continuation" - ] - }, - "config_fingerprint": "00af0bde578c1269", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "config_fingerprint": "00af0bde578c1269" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 20, - "pos": 13528066, - "epoch": 1, - "pq_idx": 20, - "rg_idx": 13528066, - "mixture": { - "cursors": { - "original": { - "file_idx": 140, - "pos": 94581760, - "epoch": 1, - "pq_idx": 140, - "rg_idx": 94581760 - }, - "midtrain_r21": { - "file_idx": 40, - "pos": 27023360, - "epoch": 1, - "pq_idx": 40, - "rg_idx": 27023360 - }, - "midtrain_r45": { - "file_idx": 20, - "pos": 13528066, - "epoch": 1, - "pq_idx": 20, - "rg_idx": 13528066 - } - }, - "cumulative_tokens": 20132675584, - "source_tokens": { - "original": 14092861440, - "midtrain_r21": 4026531840, - "midtrain_r45": 2013282304 - }, - "active_stage_idx": 2 - } - }, - "loop_state": { - "min_val_bpb": 0.7701763237939314, - "smooth_train_loss": 2.288112998721606, - "total_training_time": 164042.2579112053, - "stage_start_step": 5500, - "stage_training_flops": 98671211570213683200, - "inherited_parent_flops": 1.3236382039906714e+20, - "cumulative_pipeline_training_flops": 2.3103503196928082e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_006000.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_006000.pt deleted file mode 100644 index 39c12b0423bd99558169bfbbdd0a5b4afd06e003..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_006000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0974180f3b81e9bbfada905514b428738eff5e07bd5b2f13592956973daa97f1 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_006500.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_006500.pt deleted file mode 100644 index a123f2e532dfb528472d6efca45fb01401b497f2..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_006500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:05ef2316235989c191cdbeaf8472ad6e0762aff2a7e6d34af2b28e26c079b4b6 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_007000.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_007000.pt deleted file mode 100644 index 4bc240623f98f2ade9005a6bbf04c51d7ac79d14..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_007000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:55d8732a1c422d00c95c67ff3b0aa2c91b4ac301a19cbbcad324450853cabb26 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_007500.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_007500.pt deleted file mode 100644 index a1dcf2ba162a76d070534f78813cf323f072c590..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_007500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2fc26f44c9ec52c1295b45172a09bb5e86e5c115620cb75cdc5da028559bf55d -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_008000.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_008000.pt deleted file mode 100644 index f177bad54e882279c5753515d033418a8feb9479..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_008000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fc300d4ab2b6642c5c064b07eccf9d75e4952d1bf1a933675a5a6262bc4cfec7 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_008500.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_008500.pt deleted file mode 100644 index b1189e1c72997500d36097b255177b1107ce2fef..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_008500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d050d4332851fa3df9bae41a896b8bc037790ee2226c8a6c1b56c8ec39802a89 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_009000.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_009000.pt deleted file mode 100644 index d1adcd6e95b9dad8b5377c170a90bfae76740452..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_009000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a45cfba0cb4f10456bdaaee23bad04027bb0dab23af6f9b7614b378d6c779e23 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_009500.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_009500.pt deleted file mode 100644 index f5e34f0ad3ad250c8281c491c325a2e3ddefaba8..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_009500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:275aa51baec0db662a7866dcb1b31e69d7b32a94576ad8425bdd6b236ed33cef -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_009600.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_009600.pt deleted file mode 100644 index 4029a723594e1d7183fc8cf94cdf759c78b94f7f..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/model_009600.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:98cc4938a7f5fa9ced528fc0ca62dce7aef59c25c61d8a667ebb59832e41706c -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_006000_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_006000_rank0.pt deleted file mode 100644 index cb329d36b6037dd9f8c3b88d4c0d4beb786559d9..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_006000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:df29018d296252484de3440dea71b7e05779ce4210cf652de3c5a3bab051478a -size 11545903817 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_006500_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_006500_rank0.pt deleted file mode 100644 index bb7b8561765949fc102548ed6323ca4e9b4d7d33..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_006500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0fdd9033e42511025d80f512f5a36d1202db99a7e579261ac87e8f2f75962ade -size 11545903817 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_007000_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_007000_rank0.pt deleted file mode 100644 index 6258a338c055813a7dc25af5ac732cfcfeefd2eb..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_007000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f0c916bd5a5a7efc06f64ccaa53034f3a3fb7e3ac409a8e8305f122b4b50c74a -size 11545903817 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_007500_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_007500_rank0.pt deleted file mode 100644 index 5c548aca7567ccb79b8e9ab2e722e64325395dc4..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_007500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:47fbef0511082bc891a659193a64450dc5c60bb72dfbdb5f4d60194a6b6b4efb -size 11545903817 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_008000_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_008000_rank0.pt deleted file mode 100644 index 046081fd8c334bd1cffd0182cf9a5225ecf1147a..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_008000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:822c6493cee4683da090d11c4454c31f94c672419361361f3aeff656ef842513 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_008500_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_008500_rank0.pt deleted file mode 100644 index f11f7d10408ea2df8db13aefd148b9fae3539455..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_008500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:507e0e2264fedf4e429ad3efeb279a443e61b9311c2dca9f2efeaefa0e2b66d1 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_009000_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_009000_rank0.pt deleted file mode 100644 index 0ed48ac6e161925e62ad9ff9262ad54fb57578ce..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_009000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:91468f31292f773249bb9d6d59d1d7e5fc99c8e716efceae3157cb1953713972 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_009500_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_009500_rank0.pt deleted file mode 100644 index 2eb6c1be6cff10168583342a4ea4f099dadfe164..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_009500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:889a95c06c4543ebb315d34b9e7c6564bc8c10a2513120875399dc65149d8008 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_009600_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_009600_rank0.pt deleted file mode 100644 index 18b98014a908ef0b6aba4aa5898a6e31b4bbd5b6..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints/optim_009600_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:19dade99b1d7fabb2e6f62c2c8420f1c51f09f28eba50000a93b9f2485efa737 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/config.json b/experiments/Think.Unbounded-d32-v2mix-cont/config.json deleted file mode 100644 index 8ba3a15d1ff154c5833972e7398f69395658f7b6..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/config.json +++ /dev/null @@ -1,119 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "branch": { - "parent_experiment_id": "Think.Unbounded-d32", - "parent_step": 5500, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r21": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_21/data", - "validation_shard": 65, - "num_train_shards": 65, - "download_workers": 4 - }, - "midtrain_r45": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "5250c497c8c72558a573dbcd216e26ea3c76e6e2", - "subfolder": "mixed/v2/ratio_45/data", - "validation_shard": 33, - "num_train_shards": 33, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r21" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r45" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.02, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32-v2mix-cont", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-v2", - "0-21-45", - "midtrain-2.88ep", - "data-repair-continuation" - ] - }, - "config_fingerprint": "00af0bde578c1269", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont" -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/evals/samples.json b/experiments/Think.Unbounded-d32-v2mix-cont/evals/samples.json deleted file mode 100644 index 1b01d4fb2fb32770f9a9bbd330c32efbe1a078e2..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 9600)", - "step": 9600, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the most important city in the world. It is the capital of the most powerful" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is Au, and that of silver is Ag. The symbol of the former is Ag" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If yesterday was Saturday, then tomorrow will be Sunday. If yesterday was" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is cold, and the opposite of cold is hot. The opposite of hot is cold" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: Mercury, Venus, Mars, Jupiter, Saturn, and Herschel. The earth" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a deep, rich, velvety crimson, the same shade as the rose" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is 5.\n\nIf 3x + 4 = 13, then x" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nTHIS essay originated with Mr. W. C. Barrett, who says of it: \"Mr. Galton's 'Life and Times,' and his Life of Darwin.' \"It is mainly a half-tone reprint from Galton's (vice supra) life, to which I have added some important facts which I had not been able to procure when the original edition was published, and which bear directly upon some matters of interest.\n\nIt may, however, be as well to draw attention more particularly to the point to which Mr. Barrett refers in the above quotation, as this contains, I believe, the fundamental distinction between the Darwin", - "<|bos|>Lady Saumarez -- Scotch rose shaded white. Pheasant's Eye -- Pure white, dark eye. Marigold -- Striking light golden color. Maddereti -- A most desirable purple.\n\nRose -- -A superb full flowering rose of a most beautiful shade of deep rose. Sunrise -- Pure tints of fawn and yellow.\n\nCarnation Marguerite\n\nLarge Flowered\n\nnow offered. Much hardier and blooms more abundantly.\n\nPrice, any variety, 10 cts. each; set of 1 3 for 45 cents.\n\nEasily Makes a Good Center\n\nAnywhere.\n\n", - "<|bos|>RIDGEN, TOLLAND\n\nFRIDLAND, Dyfe, y 1665.\n\nTHE CONFUCTIONS\n\nor\n\nTHE KINGDOM OF SOLOMON.\n\nTHE FIRST BOOK OF THE KING OF SOLOMON\n\nRefers to the Four Series of the Four Dispensations, God revealing himself to His Children in the various Births of His Israel. The Messiah's birth and baptism of John, the mission of the\n\nApostles, and the baptism of Christ, and latterly the majestic progress of His power, are chronicled in the Books of Moses, of the Prophets, of the Proph ets, and the", - "<|bos|>agers have already told us, and M. Crookes has already informed us, that he found concretions of this nature in the plantlice when giving a supplementary examination to them : hence, even according to his own showing, no safe conclusion can be drawn from the relative order and distribution of these silica-precipitates. At the same time it is natural to ask whether this tardy means of elucidation, instead of becoming antiquated by the lapse of the fair period which Hibbert has now completed-and well-nigh brought to a close by a misuse of some words, may not be pressed into use - --\n\nON", - "<|bos|>D\n\nio\n\nLe\n\nthered in the middle sept i post- ater bear ried ne pro avera hing tenanted\n\nCrowne he se ouch h find\n\n\u200bLa thi seeks depone \u0f51 \u0f4f \u0f63\u0f74 \u0f49\u0f7a\u0f51 \u0f61 \u0f61 \u0f61\u0f72 \u0f64\u0f74 \u0f62 \u0f61 \u0f51 \u0f62 \u3002 \u0f61\u0f72\u0970\u0f62 \u0f51 \u0f0b \u0f62\u0f66 Atth Id\n\nDedication of the Picture in the Library of Trinity that has the most frequently called forth criticism\n\nI am unable to", - "<|bos|>PREFACE\n\nCONTENTS.\n\n1. X. About a month was being wasted at the hotel, when a kindhearted friend offered to go on ahead and engage rooms for us at the best houses the day was hot-the Southern hotels always are-on the other side, and on most of them, indeed, fair profit can be made; and the diligence kept always waiting at the station and often not in readiness for an hour after the last customer had been put in, and the engineer grumbled fearfully at this irregularity of the train, because the trains gave him a pretty lively time when there was anybody waiting there to see, and the porters sometimes began", - "<|bos|>Clarke Model\n\n222000908000 DIANER DER\n\nThe proper unture dishes\n\nWilder Damases7", - "<|bos|>ossoer 0 020 oO 000000 ee OP atte eT es ee eee en eee Bee Se SERRE Nee errre reser Res eererr ee eee cerr reeer era Nee essen re ers 2008 OO OO DO ES OO ESE HD TTI MT SPEC IC PRT APOLOC H OCDOETTES 50 Ce a OC ee DDg ee CF EIOLUS) oe eA tt } 500 een eeehey n" - ] -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/evals/val_bpb.json b/experiments/Think.Unbounded-d32-v2mix-cont/evals/val_bpb.json deleted file mode 100644 index e1d3ea5a4c47a2db494a2dea248d41b3a18cd8a5..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/evals/val_bpb.json +++ /dev/null @@ -1,94 +0,0 @@ -{ - "model": "base_model (step 9600)", - "step": 9600, - "bpb": { - "val_per_position": [ - { - "start": 0, - "end": 256, - "bpb": 0.8743052003184087 - }, - { - "start": 256, - "end": 512, - "bpb": 0.7819406932260319 - }, - { - "start": 512, - "end": 768, - "bpb": 0.7648803852009343 - }, - { - "start": 768, - "end": 1024, - "bpb": 0.7539713641978073 - }, - { - "start": 1024, - "end": 1280, - "bpb": 0.7465293364604105 - }, - { - "start": 1280, - "end": 1536, - "bpb": 0.7448722646950555 - }, - { - "start": 1536, - "end": 1792, - "bpb": 0.7425067962879133 - }, - { - "start": 1792, - "end": 2048, - "bpb": 0.7376608486711664 - }, - { - "start": 2048, - "end": 2304, - "bpb": 0.736388885937533 - }, - { - "start": 2304, - "end": 2560, - "bpb": 0.7365501564730776 - }, - { - "start": 2560, - "end": 2816, - "bpb": 0.7347149366727423 - }, - { - "start": 2816, - "end": 3072, - "bpb": 0.7327325546046755 - }, - { - "start": 3072, - "end": 3328, - "bpb": 0.7303391712119879 - }, - { - "start": 3328, - "end": 3584, - "bpb": 0.7310258026119251 - }, - { - "start": 3584, - "end": 3840, - "bpb": 0.7309421641131671 - }, - { - "start": 3840, - "end": 4096, - "bpb": 0.7292406699307448 - } - ], - "val": 0.7505380497637221 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/evals/vintage_core/filtered.json b/experiments/Think.Unbounded-d32-v2mix-cont/evals/vintage_core/filtered.json deleted file mode 100644 index 5440c8e64ee55baf07f7bf49607d46e30ddc8401..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/evals/vintage_core/filtered.json +++ /dev/null @@ -1,52 +0,0 @@ -{ - "model": "base_model (step 9600)", - "step": 9600, - "bpb": {}, - "core_metric": 0.17513085966729589, - "core_results": { - "bigbench_repeat_copy_logic": 0.0, - "copa": 0.6899999976158142, - "bigbench_operators": 0.12857143580913544, - "agi_eval_lsat_ar": 0.208695650100708, - "winograd": 0.6410256624221802, - "openbook_qa": 0.3020000159740448, - "arc_challenge": 0.2815699577331543, - "commonsense_qa": 0.24897624552249908, - "winogrande": 0.5453827977180481, - "piqa": 0.6383624076843262, - "jeopardy": 0.04212454333901405, - "arc_easy": 0.49455446004867554, - "boolq": 0.6009852290153503, - "lambada_openai": 0.41531798243522644, - "coqa": 0.2822014093399048, - "bigbench_language_identification": 0.24412153661251068, - "hellaswag_zeroshot": 0.39121130108833313, - "hellaswag": 0.39713627099990845, - "squad": 0.21311858296394348, - "bigbench_qa_wikidata": 0.407656729221344 - }, - "centered_results": { - "bigbench_repeat_copy_logic": 0.0, - "copa": 0.3799999952316284, - "bigbench_operators": 0.12857143580913544, - "agi_eval_lsat_ar": 0.010869562625884996, - "winograd": 0.28205132484436035, - "openbook_qa": 0.06933335463205974, - "arc_challenge": 0.04209327697753906, - "commonsense_qa": 0.06122030690312384, - "winogrande": 0.09076559543609619, - "piqa": 0.27672481536865234, - "jeopardy": 0.04212454333901405, - "arc_easy": 0.32607261339823407, - "boolq": -0.0784182999585126, - "lambada_openai": 0.41531798243522644, - "coqa": 0.2822014093399048, - "bigbench_language_identification": 0.16845053532729448, - "hellaswag_zeroshot": 0.18828173478444418, - "hellaswag": 0.1961816946665446, - "squad": 0.21311858296394348, - "bigbench_qa_wikidata": 0.407656729221344 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/evals/vintage_core/original.json b/experiments/Think.Unbounded-d32-v2mix-cont/evals/vintage_core/original.json deleted file mode 100644 index 0e46730ae4697c99b61a2ae0e40531dc9919e8c4..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/evals/vintage_core/original.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 9600)", - "step": 9600, - "bpb": {}, - "core_metric": 0.17197526407705976, - "core_results": { - "hellaswag_zeroshot": 0.3744274079799652, - "jeopardy": 0.032593291252851486, - "bigbench_qa_wikidata": 0.3168643116950989, - "arc_easy": 0.4583333134651184, - "arc_challenge": 0.26877132058143616, - "copa": 0.6800000071525574, - "commonsense_qa": 0.26863226294517517, - "piqa": 0.6039173007011414, - "openbook_qa": 0.2840000092983246, - "lambada_openai": 0.41432175040245056, - "hellaswag": 0.3779127597808838, - "winograd": 0.6483516693115234, - "winogrande": 0.5327545404434204, - "bigbench_dyck_languages": 0.12000000476837158, - "agi_eval_lsat_ar": 0.2695651948451996, - "bigbench_cs_algorithms": 0.39772725105285645, - "bigbench_operators": 0.12857143580913544, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.20302742719650269, - "coqa": 0.2659401297569275, - "boolq": 0.5972477197647095, - "bigbench_language_identification": 0.251800000667572 - }, - "centered_results": { - "hellaswag_zeroshot": 0.1659032106399536, - "jeopardy": 0.032593291252851486, - "bigbench_qa_wikidata": 0.3168643116950989, - "arc_easy": 0.2777777512868245, - "arc_challenge": 0.025028427441914875, - "copa": 0.36000001430511475, - "commonsense_qa": 0.08579032868146895, - "piqa": 0.20783460140228271, - "openbook_qa": 0.04533334573109945, - "lambada_openai": 0.41432175040245056, - "hellaswag": 0.17055034637451172, - "winograd": 0.2967033386230469, - "winogrande": 0.06550908088684082, - "bigbench_dyck_languages": 0.12000000476837158, - "agi_eval_lsat_ar": 0.08695649355649947, - "bigbench_cs_algorithms": 0.39772725105285645, - "bigbench_operators": 0.12857143580913544, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.20302742719650269, - "coqa": 0.2659401297569275, - "boolq": -0.05987442167181716, - "bigbench_language_identification": 0.17689769050337956 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/evals/vintage_core/restyled.json b/experiments/Think.Unbounded-d32-v2mix-cont/evals/vintage_core/restyled.json deleted file mode 100644 index b44ec34551690c0da52d368c866d8d29b1b563ba..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/evals/vintage_core/restyled.json +++ /dev/null @@ -1,52 +0,0 @@ -{ - "model": "base_model (step 9600)", - "step": 9600, - "bpb": {}, - "core_metric": 0.17594216195498452, - "core_results": { - "bigbench_repeat_copy_logic": 0.0, - "copa": 0.6800000071525574, - "bigbench_operators": 0.12857143580913544, - "agi_eval_lsat_ar": 0.208695650100708, - "winograd": 0.6556776762008667, - "openbook_qa": 0.30400002002716064, - "arc_challenge": 0.29948803782463074, - "commonsense_qa": 0.25470924377441406, - "winogrande": 0.54222571849823, - "piqa": 0.6262320280075073, - "jeopardy": 0.04639804735779762, - "arc_easy": 0.498019814491272, - "boolq": 0.605911374092102, - "lambada_openai": 0.4144062101840973, - "coqa": 0.2730679214000702, - "bigbench_language_identification": 0.24412153661251068, - "hellaswag_zeroshot": 0.39088213443756104, - "hellaswag": 0.3973008394241333, - "squad": 0.20494864881038666, - "bigbench_qa_wikidata": 0.407656729221344 - }, - "centered_results": { - "bigbench_repeat_copy_logic": 0.0, - "copa": 0.36000001430511475, - "bigbench_operators": 0.12857143580913544, - "agi_eval_lsat_ar": 0.010869562625884996, - "winograd": 0.3113553524017334, - "openbook_qa": 0.07200002670288086, - "arc_challenge": 0.06598405043284099, - "commonsense_qa": 0.06838655471801756, - "winogrande": 0.08445143699645996, - "piqa": 0.25246405601501465, - "jeopardy": 0.04639804735779762, - "arc_easy": 0.3306930859883626, - "boolq": -0.06510439434567014, - "lambada_openai": 0.4144062101840973, - "coqa": 0.2730679214000702, - "bigbench_language_identification": 0.16845053532729448, - "hellaswag_zeroshot": 0.18784284591674805, - "hellaswag": 0.19640111923217773, - "squad": 0.20494864881038666, - "bigbench_qa_wikidata": 0.407656729221344 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/evals/vintage_core/summary.csv b/experiments/Think.Unbounded-d32-v2mix-cont/evals/vintage_core/summary.csv deleted file mode 100644 index 0c099963e66a18b5c176d1b2734943efa6f02444..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/evals/vintage_core/summary.csv +++ /dev/null @@ -1,4 +0,0 @@ -bundle,native_core,common_20_core -original,0.17197526407705976,0.16328642769370433 -filtered,0.17513085966729589,0.17513085966729589 -restyled,0.17594216195498452,0.17594216195498455 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/run.json b/experiments/Think.Unbounded-d32-v2mix-cont/run.json deleted file mode 100644 index 35c23c7a7f500080cd8d1a37f7bcf4de2f5b2dea..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "branch_parent_step": 5500, - "config_fingerprint": "00af0bde578c1269", - "wandb_run_id": "2fa1b744", - "created_at": 1786217979 -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/eval_metrics.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/eval_metrics.json deleted file mode 100644 index 662802e2530b8013aa1e5748868d03d6b5d3bbb3..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/eval_metrics.json +++ /dev/null @@ -1,31 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1", - "recipe": "karpathy-discussion8", - "step": 650, - "val_bpb": 0.7357740744482647, - "min_val_bpb": 0.7357740744482647, - "per_route_bpb": {}, - "per_domain_bpb": {}, - "chatcore": { - "chatcore_metric": 0.011080512147751636, - "chatcore_cat": 0.016193069905085174, - "suite": { - "name": "karpathy", - "tasks": [ - "ARC-Easy", - "ARC-Challenge", - "MMLU", - "GSM8K", - "HumanEval" - ], - "max_generative_problems": null, - "generative_answer_format": null - }, - "ARC-Easy": 0.26641414141414144, - "ARC-Challenge": 0.26706484641638223, - "MMLU": 0.252955419455918, - "GSM8K": 0.006823351023502654, - "HumanEval": 0.0 - }, - "curriculum_summary": null -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/meta_000200.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/meta_000200.json deleted file mode 100644 index 4e505e83465096aaae371615b1bfdeb8eaf85ac1..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/meta_000200.json +++ /dev/null @@ -1,128 +0,0 @@ -{ - "step": 200, - "training_complete": false, - "val_bpb": 0.7708812390249374, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1", - "wandb_run_id": "06e3d4a8", - "wandb_group": "think-d32", - "wandb_tags": "sft,modern,karpathy-discussion8,arc,gsm8k,smoltalk,d32", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "base_step": 9600, - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "resume_from_step": null, - "experiment_id": "Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/config.json", - "parent_cumulative_flops": 2.3103503196928082e+20, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "d786bb92be8eede3edf0a4b26ded6b601b7caf3c", - "load_optimizer": 0, - "num_iterations": -1, - "num_epochs": 1, - "target_examples_per_step": 32, - "max_seq_len": null, - "device_batch_size": 2, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.02, - "warmup_ratio": 0.0, - "warmdown_ratio": 1.0, - "final_lr_frac": 0.0, - "eval_every": -1, - "eval_tokens": 20971520, - "chatcore_every": -1, - "chatcore_max_cat": -1, - "chatcore_max_sample": 32, - "save_every": 200, - "recipe": "karpathy-discussion8", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "karpathy-modern-sft-v1", - "data": { - "recipe": "karpathy-discussion8" - }, - "training": { - "num_iterations": -1, - "num_epochs": 1, - "target_examples_per_step": 32, - "load_optimizer": 0, - "device_batch_size": 2, - "init_lr_frac": 0.02, - "warmup_ratio": 0.0, - "warmdown_ratio": 1.0, - "final_lr_frac": 0.0, - "eval_every": -1, - "chatcore_every": -1, - "save_every": 200 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "modern", - "karpathy-discussion8", - "arc", - "gsm8k", - "smoltalk", - "d32" - ] - }, - "config_fingerprint": "537247050f1ad6b0", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "537247050f1ad6b0" - }, - "loop_state": { - "step": 200, - "total_training_time": 299.3075273036957, - "min_val_bpb": Infinity, - "smooth_train_loss": 2.3022586685708264, - "mfu": 13.611360314941303, - "tok_per_sec": 11730, - "stage_training_flops": 54898630603591680, - "stage_training_tokens": 4783930, - "inherited_parent_flops": 2.3103503196928082e+20, - "cumulative_pipeline_training_flops": 2.3108993059988442e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/meta_000400.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/meta_000400.json deleted file mode 100644 index 0ece18cdc920f603ab0a7b4a7c08308ab328ede7..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/meta_000400.json +++ /dev/null @@ -1,128 +0,0 @@ -{ - "step": 400, - "training_complete": false, - "val_bpb": 0.7708812390249374, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1", - "wandb_run_id": "06e3d4a8", - "wandb_group": "think-d32", - "wandb_tags": "sft,modern,karpathy-discussion8,arc,gsm8k,smoltalk,d32", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "base_step": 9600, - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "resume_from_step": null, - "experiment_id": "Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/config.json", - "parent_cumulative_flops": 2.3103503196928082e+20, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "d786bb92be8eede3edf0a4b26ded6b601b7caf3c", - "load_optimizer": 0, - "num_iterations": -1, - "num_epochs": 1, - "target_examples_per_step": 32, - "max_seq_len": null, - "device_batch_size": 2, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.02, - "warmup_ratio": 0.0, - "warmdown_ratio": 1.0, - "final_lr_frac": 0.0, - "eval_every": -1, - "eval_tokens": 20971520, - "chatcore_every": -1, - "chatcore_max_cat": -1, - "chatcore_max_sample": 32, - "save_every": 200, - "recipe": "karpathy-discussion8", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "karpathy-modern-sft-v1", - "data": { - "recipe": "karpathy-discussion8" - }, - "training": { - "num_iterations": -1, - "num_epochs": 1, - "target_examples_per_step": 32, - "load_optimizer": 0, - "device_batch_size": 2, - "init_lr_frac": 0.02, - "warmup_ratio": 0.0, - "warmdown_ratio": 1.0, - "final_lr_frac": 0.0, - "eval_every": -1, - "chatcore_every": -1, - "save_every": 200 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "modern", - "karpathy-discussion8", - "arc", - "gsm8k", - "smoltalk", - "d32" - ] - }, - "config_fingerprint": "537247050f1ad6b0", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "537247050f1ad6b0" - }, - "loop_state": { - "step": 400, - "total_training_time": 618.5809240341187, - "min_val_bpb": Infinity, - "smooth_train_loss": 1.6475925896047963, - "mfu": 16.554368928120176, - "tok_per_sec": 14266, - "stage_training_flops": 110296864416669696, - "stage_training_tokens": 9611396, - "inherited_parent_flops": 2.3103503196928082e+20, - "cumulative_pipeline_training_flops": 2.3114532883369748e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/meta_000600.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/meta_000600.json deleted file mode 100644 index 916a4e25a4f40b79ff9c2f0f7dee4aeb7496fb31..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/meta_000600.json +++ /dev/null @@ -1,128 +0,0 @@ -{ - "step": 600, - "training_complete": false, - "val_bpb": 0.7708812390249374, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1", - "wandb_run_id": "06e3d4a8", - "wandb_group": "think-d32", - "wandb_tags": "sft,modern,karpathy-discussion8,arc,gsm8k,smoltalk,d32", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "base_step": 9600, - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "resume_from_step": null, - "experiment_id": "Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/config.json", - "parent_cumulative_flops": 2.3103503196928082e+20, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "d786bb92be8eede3edf0a4b26ded6b601b7caf3c", - "load_optimizer": 0, - "num_iterations": -1, - "num_epochs": 1, - "target_examples_per_step": 32, - "max_seq_len": null, - "device_batch_size": 2, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.02, - "warmup_ratio": 0.0, - "warmdown_ratio": 1.0, - "final_lr_frac": 0.0, - "eval_every": -1, - "eval_tokens": 20971520, - "chatcore_every": -1, - "chatcore_max_cat": -1, - "chatcore_max_sample": 32, - "save_every": 200, - "recipe": "karpathy-discussion8", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "karpathy-modern-sft-v1", - "data": { - "recipe": "karpathy-discussion8" - }, - "training": { - "num_iterations": -1, - "num_epochs": 1, - "target_examples_per_step": 32, - "load_optimizer": 0, - "device_batch_size": 2, - "init_lr_frac": 0.02, - "warmup_ratio": 0.0, - "warmdown_ratio": 1.0, - "final_lr_frac": 0.0, - "eval_every": -1, - "chatcore_every": -1, - "save_every": 200 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "modern", - "karpathy-discussion8", - "arc", - "gsm8k", - "smoltalk", - "d32" - ] - }, - "config_fingerprint": "537247050f1ad6b0", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "537247050f1ad6b0" - }, - "loop_state": { - "step": 600, - "total_training_time": 934.3550770282745, - "min_val_bpb": Infinity, - "smooth_train_loss": 1.8279502507943437, - "mfu": 18.70212930901884, - "tok_per_sec": 16117, - "stage_training_flops": 165484589196423168, - "stage_training_tokens": 14420518, - "inherited_parent_flops": 2.3103503196928082e+20, - "cumulative_pipeline_training_flops": 2.3120051655847723e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/meta_000650.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/meta_000650.json deleted file mode 100644 index 4555e6abffc8699a4005e1f0411b0ed2bdc9e920..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/meta_000650.json +++ /dev/null @@ -1,128 +0,0 @@ -{ - "step": 650, - "training_complete": true, - "val_bpb": 0.7357740744482647, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1", - "wandb_run_id": "06e3d4a8", - "wandb_group": "think-d32", - "wandb_tags": "sft,modern,karpathy-discussion8,arc,gsm8k,smoltalk,d32", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "base_step": 9600, - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "resume_from_step": null, - "experiment_id": "Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/config.json", - "parent_cumulative_flops": 2.3103503196928082e+20, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "d786bb92be8eede3edf0a4b26ded6b601b7caf3c", - "load_optimizer": 0, - "num_iterations": -1, - "num_epochs": 1, - "target_examples_per_step": 32, - "max_seq_len": null, - "device_batch_size": 2, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.02, - "warmup_ratio": 0.0, - "warmdown_ratio": 1.0, - "final_lr_frac": 0.0, - "eval_every": -1, - "eval_tokens": 20971520, - "chatcore_every": -1, - "chatcore_max_cat": -1, - "chatcore_max_sample": 32, - "save_every": 200, - "recipe": "karpathy-discussion8", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "karpathy-modern-sft-v1", - "data": { - "recipe": "karpathy-discussion8" - }, - "training": { - "num_iterations": -1, - "num_epochs": 1, - "target_examples_per_step": 32, - "load_optimizer": 0, - "device_batch_size": 2, - "init_lr_frac": 0.02, - "warmup_ratio": 0.0, - "warmdown_ratio": 1.0, - "final_lr_frac": 0.0, - "eval_every": -1, - "chatcore_every": -1, - "save_every": 200 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "modern", - "karpathy-discussion8", - "arc", - "gsm8k", - "smoltalk", - "d32" - ] - }, - "config_fingerprint": "537247050f1ad6b0", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "537247050f1ad6b0" - }, - "loop_state": { - "step": 650, - "total_training_time": 1013.77108335495, - "min_val_bpb": 0.7357740744482647, - "smooth_train_loss": 1.819000634696336, - "mfu": 18.620771423739047, - "tok_per_sec": 16047, - "stage_training_flops": 179443894877134848, - "stage_training_tokens": 15636948, - "inherited_parent_flops": 2.3103503196928082e+20, - "cumulative_pipeline_training_flops": 2.3121447586415795e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/model_000200.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/model_000200.pt deleted file mode 100644 index ac72cee76e8a1de837bc8713681af8fee2533340..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/model_000200.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:be723876bf78e27ea88118195e47b5ece3b9bf991fde1b94136150008c86a53d -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/model_000400.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/model_000400.pt deleted file mode 100644 index a297c384b6a4c898b8b4fb09de76e6f26ebfb740..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/model_000400.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:cd78dd298f892c2caf1cce1d42ca9b08a9e19c5c5bd66c8e26c873efb49b9ed3 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/model_000600.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/model_000600.pt deleted file mode 100644 index 2e55d666ddcea31689283c90c381326f24c609fd..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/model_000600.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f64cc1da0a0ea9407de11a7ef70d011a5811775d60377b5c825ea1b2062a282d -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/model_000650.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/model_000650.pt deleted file mode 100644 index 04e150790732658d2133307dac8d998fd6780a79..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/model_000650.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e73543c15827e67fb1880bf5b1f629302b80981a9f876c32728675b13f1494f0 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/optim_000200_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/optim_000200_rank0.pt deleted file mode 100644 index da7eb0600c6f8dd529f6bdf3c7d8f0b9a4a85d1d..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/optim_000200_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:838b9ee4885d7647e5f792338257d6777e839f05d7fa3b35609b5aeed1fe7893 -size 11545903753 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/optim_000400_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/optim_000400_rank0.pt deleted file mode 100644 index c264743e2974ea0133665aaa045500c44cd0166a..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/optim_000400_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fee93c6c8a837448f25680423872b09f1cf550ce327759dd7868262a1a966114 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/optim_000600_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/optim_000600_rank0.pt deleted file mode 100644 index fd52d1d2d8c794de13e74bd7d1ab4047d661bdcb..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/optim_000600_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:974eb6c2f3d32abd916531b817b6dca2b7f1b6a5adaea94a6778254fb24d9ca8 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/optim_000650_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/optim_000650_rank0.pt deleted file mode 100644 index ee25b53725387b4ac8ff39d7f42d7483bafdcc1f..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/checkpoints/optim_000650_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:acd58691ac0598abd6ea87984d1a8e955a10a71f36f40de11e10e0c77934b303 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/config.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/config.json deleted file mode 100644 index a6736c0375569b003fc32dbb830539259e7fadd8..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/config.json +++ /dev/null @@ -1,42 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "karpathy-modern-sft-v1", - "data": { - "recipe": "karpathy-discussion8" - }, - "training": { - "num_iterations": -1, - "num_epochs": 1, - "target_examples_per_step": 32, - "load_optimizer": 0, - "device_batch_size": 2, - "init_lr_frac": 0.02, - "warmup_ratio": 0.0, - "warmdown_ratio": 1.0, - "final_lr_frac": 0.0, - "eval_every": -1, - "chatcore_every": -1, - "save_every": 200 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "modern", - "karpathy-discussion8", - "arc", - "gsm8k", - "smoltalk", - "d32" - ] - }, - "config_fingerprint": "537247050f1ad6b0", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1" -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/evals/chatcore.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/evals/chatcore.json deleted file mode 100644 index 709b5f405edea91f0ec18533345bc1c6e1b86365..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/evals/chatcore.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "stage": "sft", - "step": 650, - "total_training_flops": 2.3121447586415795e+20, - "stage_training_flops": 1.7944389487713485e+17, - "inherited_parent_flops": 2.3103503196928082e+20, - "cumulative_pipeline_training_flops": 2.3121447586415795e+20, - "results": { - "ARC-Easy": 0.26641414141414144, - "ARC-Challenge": 0.26706484641638223, - "MMLU": 0.252955419455918, - "GSM8K": 0.006823351023502654, - "HumanEval": 0.0 - }, - "chatcore_metric": 0.011080512147751636, - "chatcore_suite": { - "name": "karpathy", - "tasks": [ - "ARC-Easy", - "ARC-Challenge", - "MMLU", - "GSM8K", - "HumanEval" - ], - "max_generative_problems": null, - "generative_answer_format": null - }, - "complete": true -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/evals/vintage_vs_modern_karpathy.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/evals/vintage_vs_modern_karpathy.json deleted file mode 100644 index 5577338123986d100c111cd6692a86eaef718856..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/evals/vintage_vs_modern_karpathy.json +++ /dev/null @@ -1,52 +0,0 @@ -{ - "schema_version": 1, - "generated_at": 1786862658, - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_step": 9600, - "suite": { - "name": "karpathy", - "tasks": [ - "ARC-Easy", - "ARC-Challenge", - "MMLU", - "GSM8K", - "HumanEval" - ], - "max_generative_problems": null, - "generative_answer_format": null - }, - "vintage": { - "experiment_id": "Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2", - "step": 42, - "chatcore_metric": -0.011321608241655831, - "results": { - "ARC-Easy": 0.24579124579124578, - "ARC-Challenge": 0.23037542662116042, - "MMLU": 0.23137729668138443, - "GSM8K": 0.0, - "HumanEval": 0.0 - } - }, - "modern": { - "experiment_id": "Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1", - "step": 650, - "chatcore_metric": 0.011080512147751636, - "results": { - "ARC-Easy": 0.26641414141414144, - "ARC-Challenge": 0.26706484641638223, - "MMLU": 0.252955419455918, - "GSM8K": 0.006823351023502654, - "HumanEval": 0.0 - } - }, - "modern_minus_vintage": { - "chatcore_metric": 0.02240212038940747, - "results": { - "ARC-Easy": 0.020622895622895654, - "ARC-Challenge": 0.036689419795221806, - "MMLU": 0.021578122774533554, - "GSM8K": 0.006823351023502654, - "HumanEval": 0.0 - } - } -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/run.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/run.json deleted file mode 100644 index 4ded934e5e8e5c403a98debdc3620c686be77625..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1", - "stage": "sft", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_checkpoint_step": 9600, - "branch_parent_step": null, - "config_fingerprint": "537247050f1ad6b0", - "wandb_run_id": "06e3d4a8", - "created_at": 1786843832 -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/summary.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/summary.json deleted file mode 100644 index b35d5c78263451d8fe564ca8fd6cad98955fa3e8..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1/summary.json +++ /dev/null @@ -1,25 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1", - "stage": "sft", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_checkpoint_step": 9600, - "step": 650, - "stage_training_flops": 1.7944389487713485e+17, - "inherited_parent_flops": 2.3103503196928082e+20, - "cumulative_pipeline_training_flops": 2.3121447586415795e+20, - "chatcore_metric": 0.011080512147751636, - "chat_results": { - "ARC-Easy": 0.26641414141414144, - "ARC-Challenge": 0.26706484641638223, - "MMLU": 0.252955419455918, - "GSM8K": 0.006823351023502654, - "HumanEval": 0.0 - }, - "training_time_seconds": 1013.77108335495, - "config_fingerprint": "537247050f1ad6b0", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "d786bb92be8eede3edf0a4b26ded6b601b7caf3c", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/06e3d4a8", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-karpathy-modern-sft-v1" -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints/eval_metrics.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints/eval_metrics.json deleted file mode 100644 index 404b8139fbaa9105438e7e32075a7226c55bb429..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints/eval_metrics.json +++ /dev/null @@ -1,143 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1", - "recipe": "nanochat-default", - "step": 160, - "val_bpb": 0.32778816957977247, - "min_val_bpb": 0.32778816957977247, - "per_route_bpb": {}, - "per_domain_bpb": {}, - "chatcore": { - "chatcore_metric": 0.13532147122389065, - "chatcore_cat": 0.19185974353312688, - "suite": { - "name": "karpathy", - "tasks": [ - "ARC-Easy", - "ARC-Challenge", - "MMLU", - "GSM8K", - "HumanEval" - ], - "max_generative_problems": null, - "generative_answer_format": null - }, - "ARC-Easy": 0.4553872053872054, - "ARC-Challenge": 0.378839590443686, - "MMLU": 0.3474576271186441, - "GSM8K": 0.06444275966641395, - "HumanEval": 0.036585365853658534 - }, - "curriculum_summary": null, - "sft_mixture_summary": { - "schema_version": 1, - "recipe": "nanochat-default", - "selection": { - "algorithm": "TaskMixture(seed=42) deterministic global shuffle then prefix", - "selected_presentations": 652950, - "full_mixture_presentations": 1071759, - "selection_sha256": "975f61958382acd3a3cd9b01c2f3204cff8982908365503ac8972e29d38bfec2" - }, - "sources": { - "smoltalk": { - "kind": "huggingface", - "repo": "HuggingFaceTB/smol-smoltalk", - "split": "train", - "dataset_fingerprint": "7776c4bbcd5d614f", - "configured_repetitions": 1, - "available_rows": 460341, - "resolved_revision": "f73fe857d519ff6ac5af2ea67c4d3834da7b8bcc", - "full_mixture_presentations": 460341, - "selected_presentations": 280414, - "selected_presentations_by_replica": [ - 280414 - ], - "selected_distinct_rows": 280414, - "selected_fraction": 0.4294570794088368 - }, - "identity": { - "kind": "jsonl", - "url": "https://karpathy-public.s3.us-west-2.amazonaws.com/identity_conversations.jsonl", - "sha256": "85a6a96342b88addeb1b8267933c47a6e3ceb1dfc07b305b805fd7ff8106878e", - "configured_repetitions": 2, - "available_rows": 1000, - "full_mixture_presentations": 2000, - "selected_presentations": 1241, - "selected_presentations_by_replica": [ - 630, - 611 - ], - "selected_distinct_rows": 874, - "selected_fraction": 0.0019006049467799986 - }, - "mmlu": { - "kind": "huggingface", - "repo": "cais/mmlu", - "subset": "all", - "split": "auxiliary_train", - "dataset_fingerprint": "5e98d840a8432dc6", - "configured_repetitions": 3, - "available_rows": 99842, - "resolved_revision": "c30699e8356da336a370243923dbaf21066bb9fe", - "full_mixture_presentations": 299526, - "selected_presentations": 182570, - "selected_presentations_by_replica": [ - 60835, - 60932, - 60803 - ], - "selected_distinct_rows": 93775, - "selected_fraction": 0.2796079332261276 - }, - "gsm8k": { - "kind": "huggingface", - "repo": "openai/gsm8k", - "subset": "main", - "split": "train", - "dataset_fingerprint": "8b93a690dcf7094e", - "configured_repetitions": 4, - "available_rows": 7473, - "resolved_revision": "740312add88f781978c0658806c59bc2815b9866", - "full_mixture_presentations": 29892, - "selected_presentations": 18182, - "selected_presentations_by_replica": [ - 4516, - 4549, - 4540, - 4577 - ], - "selected_distinct_rows": 7288, - "selected_fraction": 0.02784593000995482 - }, - "simple_spelling": { - "kind": "deterministic_generated", - "generator": "tasks.spellingbee.SimpleSpelling", - "word_list_url": "https://raw.githubusercontent.com/dwyl/english-words/refs/heads/master/words_alpha.txt", - "word_list_sha256": "643c7f71d5caee9806be56f5ed83ba36e2246a06a04bcf4aaa0576790bb90d66", - "configured_repetitions": 1, - "available_rows": 200000, - "full_mixture_presentations": 200000, - "selected_presentations": 121661, - "selected_presentations_by_replica": [ - 121661 - ], - "selected_distinct_rows": 121661, - "selected_fraction": 0.18632513975036374 - }, - "spelling_bee": { - "kind": "deterministic_generated", - "generator": "tasks.spellingbee.SpellingBee", - "word_list_url": "https://raw.githubusercontent.com/dwyl/english-words/refs/heads/master/words_alpha.txt", - "word_list_sha256": "643c7f71d5caee9806be56f5ed83ba36e2246a06a04bcf4aaa0576790bb90d66", - "configured_repetitions": 1, - "available_rows": 80000, - "full_mixture_presentations": 80000, - "selected_presentations": 48882, - "selected_presentations_by_replica": [ - 48882 - ], - "selected_distinct_rows": 48882, - "selected_fraction": 0.07486331265793705 - } - } - } -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints/meta_000160.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints/meta_000160.json deleted file mode 100644 index ed40286e3403b2f17be7ca2978adccfb3940c66e..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints/meta_000160.json +++ /dev/null @@ -1,250 +0,0 @@ -{ - "step": 160, - "training_complete": true, - "val_bpb": 0.32778816957977247, - "total_batch_size": 2097152, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1", - "wandb_run_id": "8c553682", - "wandb_group": "think-d32", - "wandb_tags": "sft,modern,nanochat-default,data-matched,compute-controlled,d32", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "base_step": 9600, - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "resume_from_step": null, - "experiment_id": "Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/config.json", - "parent_cumulative_flops": 2.3103503196928082e+20, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "3c34775d260a016a79ab46ea76f8c8e48458fd68", - "load_optimizer": 0, - "num_iterations": -1, - "num_epochs": 1, - "target_examples_per_step": 0, - "max_seq_len": null, - "device_batch_size": 2, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.03, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": -1, - "eval_tokens": 20971520, - "chatcore_every": -1, - "chatcore_max_cat": -1, - "chatcore_max_sample": 32, - "save_every": -1, - "recipe": "nanochat-default", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "max_train_presentations": 652950, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "nanochat-default-datamatch-v1", - "data": { - "recipe": "nanochat-default", - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "max_train_presentations": 652950 - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 2, - "warmup_ratio": 0.03, - "warmdown_ratio": 0.5, - "init_lr_frac": 0.8, - "final_lr_frac": 0.0, - "eval_every": -1, - "chatcore_every": -1, - "save_every": -1 - }, - "comparison": { - "reference_sft": "pre1930-curriculum-c3-robust-v2", - "reference_train_presentations": 652950, - "reference_optimizer_steps": 42, - "reference_stage_training_flops": 1.0107782648656036e+18, - "matching_priority": "row_presentations", - "note": "Realized compute is measured because conversation lengths differ." - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "modern", - "nanochat-default", - "data-matched", - "compute-controlled", - "d32" - ] - }, - "config_fingerprint": "6359be7150ece075", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6359be7150ece075" - }, - "sft_mixture_summary": { - "schema_version": 1, - "recipe": "nanochat-default", - "selection": { - "algorithm": "TaskMixture(seed=42) deterministic global shuffle then prefix", - "selected_presentations": 652950, - "full_mixture_presentations": 1071759, - "selection_sha256": "975f61958382acd3a3cd9b01c2f3204cff8982908365503ac8972e29d38bfec2" - }, - "sources": { - "smoltalk": { - "kind": "huggingface", - "repo": "HuggingFaceTB/smol-smoltalk", - "split": "train", - "dataset_fingerprint": "7776c4bbcd5d614f", - "configured_repetitions": 1, - "available_rows": 460341, - "resolved_revision": "f73fe857d519ff6ac5af2ea67c4d3834da7b8bcc", - "full_mixture_presentations": 460341, - "selected_presentations": 280414, - "selected_presentations_by_replica": [ - 280414 - ], - "selected_distinct_rows": 280414, - "selected_fraction": 0.4294570794088368 - }, - "identity": { - "kind": "jsonl", - "url": "https://karpathy-public.s3.us-west-2.amazonaws.com/identity_conversations.jsonl", - "sha256": "85a6a96342b88addeb1b8267933c47a6e3ceb1dfc07b305b805fd7ff8106878e", - "configured_repetitions": 2, - "available_rows": 1000, - "full_mixture_presentations": 2000, - "selected_presentations": 1241, - "selected_presentations_by_replica": [ - 630, - 611 - ], - "selected_distinct_rows": 874, - "selected_fraction": 0.0019006049467799986 - }, - "mmlu": { - "kind": "huggingface", - "repo": "cais/mmlu", - "subset": "all", - "split": "auxiliary_train", - "dataset_fingerprint": "5e98d840a8432dc6", - "configured_repetitions": 3, - "available_rows": 99842, - "resolved_revision": "c30699e8356da336a370243923dbaf21066bb9fe", - "full_mixture_presentations": 299526, - "selected_presentations": 182570, - "selected_presentations_by_replica": [ - 60835, - 60932, - 60803 - ], - "selected_distinct_rows": 93775, - "selected_fraction": 0.2796079332261276 - }, - "gsm8k": { - "kind": "huggingface", - "repo": "openai/gsm8k", - "subset": "main", - "split": "train", - "dataset_fingerprint": "8b93a690dcf7094e", - "configured_repetitions": 4, - "available_rows": 7473, - "resolved_revision": "740312add88f781978c0658806c59bc2815b9866", - "full_mixture_presentations": 29892, - "selected_presentations": 18182, - "selected_presentations_by_replica": [ - 4516, - 4549, - 4540, - 4577 - ], - "selected_distinct_rows": 7288, - "selected_fraction": 0.02784593000995482 - }, - "simple_spelling": { - "kind": "deterministic_generated", - "generator": "tasks.spellingbee.SimpleSpelling", - "word_list_url": "https://raw.githubusercontent.com/dwyl/english-words/refs/heads/master/words_alpha.txt", - "word_list_sha256": "643c7f71d5caee9806be56f5ed83ba36e2246a06a04bcf4aaa0576790bb90d66", - "configured_repetitions": 1, - "available_rows": 200000, - "full_mixture_presentations": 200000, - "selected_presentations": 121661, - "selected_presentations_by_replica": [ - 121661 - ], - "selected_distinct_rows": 121661, - "selected_fraction": 0.18632513975036374 - }, - "spelling_bee": { - "kind": "deterministic_generated", - "generator": "tasks.spellingbee.SpellingBee", - "word_list_url": "https://raw.githubusercontent.com/dwyl/english-words/refs/heads/master/words_alpha.txt", - "word_list_sha256": "643c7f71d5caee9806be56f5ed83ba36e2246a06a04bcf4aaa0576790bb90d66", - "configured_repetitions": 1, - "available_rows": 80000, - "full_mixture_presentations": 80000, - "selected_presentations": 48882, - "selected_presentations_by_replica": [ - 48882 - ], - "selected_distinct_rows": 48882, - "selected_fraction": 0.07486331265793705 - } - } - }, - "loop_state": { - "step": 160, - "total_training_time": 6951.126858472824, - "min_val_bpb": 0.32778816957977247, - "smooth_train_loss": 0.8890764356615257, - "mfu": 52.42433466002523, - "tok_per_sec": 45180, - "stage_training_flops": 3.8505838661546803e+18, - "stage_training_tokens": 335544320, - "inherited_parent_flops": 2.3103503196928082e+20, - "cumulative_pipeline_training_flops": 2.348856158354355e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints/model_000160.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints/model_000160.pt deleted file mode 100644 index 760e6cc7f577201d06929b2060ceb3930e4469d0..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints/model_000160.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2399dcc7f5c4f7974302eef77e4cec0e3ae67c352f1f467c3e33428df015f400 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints/optim_000160_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints/optim_000160_rank0.pt deleted file mode 100644 index ac52a39ab4b5d43cdbe9e4e1ee6c0efeb0b13edf..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/checkpoints/optim_000160_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:594ea9cb75c35bef2d8d8fc4eece625a41d85fefdce9efeffe9917a0d285a0d9 -size 11545903753 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/config.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/config.json deleted file mode 100644 index 96af1cacb2bfe161c18a1f762cc2320fbb84683c..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/config.json +++ /dev/null @@ -1,50 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "nanochat-default-datamatch-v1", - "data": { - "recipe": "nanochat-default", - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "max_train_presentations": 652950 - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 2, - "warmup_ratio": 0.03, - "warmdown_ratio": 0.5, - "init_lr_frac": 0.8, - "final_lr_frac": 0.0, - "eval_every": -1, - "chatcore_every": -1, - "save_every": -1 - }, - "comparison": { - "reference_sft": "pre1930-curriculum-c3-robust-v2", - "reference_train_presentations": 652950, - "reference_optimizer_steps": 42, - "reference_stage_training_flops": 1.0107782648656036e+18, - "matching_priority": "row_presentations", - "note": "Realized compute is measured because conversation lengths differ." - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "modern", - "nanochat-default", - "data-matched", - "compute-controlled", - "d32" - ] - }, - "config_fingerprint": "6359be7150ece075", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1" -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/evals/chatcore.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/evals/chatcore.json deleted file mode 100644 index eb1e7be7c3bbf64539aa7c24908f14b02d1d2eac..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/evals/chatcore.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "stage": "sft", - "step": 160, - "total_training_flops": 2.348856158354355e+20, - "stage_training_flops": 3.8505838661546803e+18, - "inherited_parent_flops": 2.3103503196928082e+20, - "cumulative_pipeline_training_flops": 2.348856158354355e+20, - "results": { - "ARC-Easy": 0.4553872053872054, - "ARC-Challenge": 0.378839590443686, - "MMLU": 0.3474576271186441, - "GSM8K": 0.06444275966641395, - "HumanEval": 0.036585365853658534 - }, - "chatcore_metric": 0.13532147122389065, - "chatcore_suite": { - "name": "karpathy", - "tasks": [ - "ARC-Easy", - "ARC-Challenge", - "MMLU", - "GSM8K", - "HumanEval" - ], - "max_generative_problems": null, - "generative_answer_format": null - }, - "complete": true -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/evals/vintage_vs_modern_datamatch.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/evals/vintage_vs_modern_datamatch.json deleted file mode 100644 index 88d157d08eb92bc5ce643c69abcf89a32c3de270..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/evals/vintage_vs_modern_datamatch.json +++ /dev/null @@ -1,57 +0,0 @@ -{ - "schema_version": 1, - "generated_at": 1786875728, - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_step": 9600, - "matching": { - "priority": "row_presentations", - "target_presentations": 652950, - "reference_stage_training_flops": 1.0107782648656036e+18 - }, - "suite": { - "name": "karpathy", - "tasks": [ - "ARC-Easy", - "ARC-Challenge", - "MMLU", - "GSM8K", - "HumanEval" - ], - "max_generative_problems": null, - "generative_answer_format": null - }, - "vintage": { - "experiment_id": "Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2", - "step": 42, - "chatcore_metric": -0.011321608241655831, - "results": { - "ARC-Easy": 0.24579124579124578, - "ARC-Challenge": 0.23037542662116042, - "MMLU": 0.23137729668138443, - "GSM8K": 0.0, - "HumanEval": 0.0 - } - }, - "modern_data_matched": { - "experiment_id": "Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1", - "step": 160, - "chatcore_metric": 0.13532147122389065, - "results": { - "ARC-Easy": 0.4553872053872054, - "ARC-Challenge": 0.378839590443686, - "MMLU": 0.3474576271186441, - "GSM8K": 0.06444275966641395, - "HumanEval": 0.036585365853658534 - } - }, - "modern_minus_vintage": { - "chatcore_metric": 0.1466430794655465, - "results": { - "ARC-Easy": 0.2095959595959596, - "ARC-Challenge": 0.14846416382252556, - "MMLU": 0.11608033043725965, - "GSM8K": 0.06444275966641395, - "HumanEval": 0.036585365853658534 - } - } -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/run.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/run.json deleted file mode 100644 index ea6bbc17ac8661de2274ac716d8196b6b38a4364..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1", - "stage": "sft", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_checkpoint_step": 9600, - "branch_parent_step": null, - "config_fingerprint": "6359be7150ece075", - "wandb_run_id": "8c553682", - "created_at": 1786863324 -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/summary.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/summary.json deleted file mode 100644 index a757a8d3f794cc2971d9b336ff29198b06062e77..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1/summary.json +++ /dev/null @@ -1,25 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1", - "stage": "sft", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_checkpoint_step": 9600, - "step": 160, - "stage_training_flops": 3.8505838661546803e+18, - "inherited_parent_flops": 2.3103503196928082e+20, - "cumulative_pipeline_training_flops": 2.348856158354355e+20, - "chatcore_metric": 0.13532147122389065, - "chat_results": { - "ARC-Easy": 0.4553872053872054, - "ARC-Challenge": 0.378839590443686, - "MMLU": 0.3474576271186441, - "GSM8K": 0.06444275966641395, - "HumanEval": 0.036585365853658534 - }, - "training_time_seconds": 6951.126858472824, - "config_fingerprint": "6359be7150ece075", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "3c34775d260a016a79ab46ea76f8c8e48458fd68", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/8c553682", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-nanochat-default-datamatch-v1" -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints/eval_metrics.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints/eval_metrics.json deleted file mode 100644 index f3e7eb210d99f103c00dafd9da0415b5c1dc72c7..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints/eval_metrics.json +++ /dev/null @@ -1,31 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2", - "recipe": "curriculum", - "step": 42, - "val_bpb": 0.7037352212342954, - "min_val_bpb": 0.7037352212342954, - "per_route_bpb": {}, - "per_domain_bpb": {}, - "curriculum_summary": null, - "chatcore": { - "chatcore_metric": -0.011321608241655831, - "chatcore_cat": -0.018869347069426386, - "suite": { - "name": "karpathy", - "tasks": [ - "ARC-Easy", - "ARC-Challenge", - "MMLU", - "GSM8K", - "HumanEval" - ], - "max_generative_problems": null, - "generative_answer_format": null - }, - "ARC-Easy": 0.24579124579124578, - "ARC-Challenge": 0.23037542662116042, - "MMLU": 0.23137729668138443, - "GSM8K": 0.0, - "HumanEval": 0.0 - } -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints/meta_000042.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints/meta_000042.json deleted file mode 100644 index d85126cc828cf0b36d544c33a4d4f59016e00897..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints/meta_000042.json +++ /dev/null @@ -1,178 +0,0 @@ -{ - "step": 42, - "training_complete": true, - "val_bpb": 0.7037352212342954, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2", - "wandb_run_id": "036b1c71", - "wandb_group": "think-d32", - "wandb_tags": "sft,curriculum,c3,scale-max,staged,robustness,noise,d32", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "base_step": 9600, - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "resume_from_step": null, - "experiment_id": "Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/config.json", - "parent_cumulative_flops": 2.3103503196928082e+20, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "04bb043ad61d4db3e9022c422302df7b8f2dc0c9", - "load_optimizer": 0, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 2, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.03, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": 400, - "eval_tokens": 20971520, - "chatcore_every": 100000, - "chatcore_max_cat": -1, - "chatcore_max_sample": 32, - "save_every": -1, - "recipe": "curriculum", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c3-robust-v2", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C3Rv2", - "mode": "staged", - "threshold_default": 80, - "stages": [ - { - "routes": [ - "knowledge_qa" - ], - "authentic": "single" - }, - { - "routes": [ - "reasoning_qa", - "stem_reasoning", - "how_to_qa", - "opinion_qa", - "composition_qa", - "verse_qa" - ], - "calibration_qa": true - }, - { - "routes": [ - "multiturn_qa", - "narrative_grounded", - "narrative_fiction" - ], - "authentic": "multi" - } - ], - "noise": { - "rate": 0.3 - }, - "robustness": { - "stage": 0, - "epochs": 1, - "routes": { - "conversation_qa": { - "count": null - }, - "unparseable_qa": { - "count": null - }, - "typo_qa": { - "count": null - }, - "era_qa": { - "count": null - }, - "conversation_multiturn": { - "count": null - } - } - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 2, - "warmup_ratio": 0.03, - "eval_every": 400, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "curriculum", - "c3", - "scale-max", - "staged", - "robustness", - "noise", - "d32" - ] - }, - "config_fingerprint": "4462a389dc98cd5d", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "4462a389dc98cd5d" - }, - "loop_state": { - "step": 42, - "total_training_time": 1503.5729036331177, - "min_val_bpb": 0.7037352212342954, - "smooth_train_loss": 1.8593226867746961, - "mfu": 51.82901425419851, - "tok_per_sec": 44667, - "stage_training_flops": 1.0107782648656036e+18, - "inherited_parent_flops": 2.3103503196928082e+20, - "cumulative_pipeline_training_flops": 2.3204581023414642e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints/model_000042.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints/model_000042.pt deleted file mode 100644 index 0c551ed82c9b4b1612a631df6e091a98d2dcbad5..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints/model_000042.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f6a9bc83e3853f0a7ad8b27cc45388d727cc5976837f514d61ea59f80d7f0348 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints/optim_000042_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints/optim_000042_rank0.pt deleted file mode 100644 index 5ba76a7fff00118ac728cfbc451ad5a80e034586..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/checkpoints/optim_000042_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:34c0871ec1aed5ced55a463ea1b827e4ef80e286735cc42512ced11f8c579fe4 -size 11545903753 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/config.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/config.json deleted file mode 100644 index 9e02c2e96ea39a561695c7897f006221a2c6bcbb..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/config.json +++ /dev/null @@ -1,95 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c3-robust-v2", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C3Rv2", - "mode": "staged", - "threshold_default": 80, - "stages": [ - { - "routes": [ - "knowledge_qa" - ], - "authentic": "single" - }, - { - "routes": [ - "reasoning_qa", - "stem_reasoning", - "how_to_qa", - "opinion_qa", - "composition_qa", - "verse_qa" - ], - "calibration_qa": true - }, - { - "routes": [ - "multiturn_qa", - "narrative_grounded", - "narrative_fiction" - ], - "authentic": "multi" - } - ], - "noise": { - "rate": 0.3 - }, - "robustness": { - "stage": 0, - "epochs": 1, - "routes": { - "conversation_qa": { - "count": null - }, - "unparseable_qa": { - "count": null - }, - "typo_qa": { - "count": null - }, - "era_qa": { - "count": null - }, - "conversation_multiturn": { - "count": null - } - } - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 2, - "warmup_ratio": 0.03, - "eval_every": 400, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "curriculum", - "c3", - "scale-max", - "staged", - "robustness", - "noise", - "d32" - ] - }, - "config_fingerprint": "4462a389dc98cd5d", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2" -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/evals/chatcore.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/evals/chatcore.json deleted file mode 100644 index 22aaa6fbaa4fafd91f83241a19976f553ed72519..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/evals/chatcore.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "stage": "sft", - "step": 42, - "total_training_flops": 2.3204581023414642e+20, - "stage_training_flops": 1.0107782648656036e+18, - "inherited_parent_flops": 2.3103503196928082e+20, - "cumulative_pipeline_training_flops": 2.3204581023414642e+20, - "results": { - "ARC-Easy": 0.24579124579124578, - "ARC-Challenge": 0.23037542662116042, - "MMLU": 0.23137729668138443, - "GSM8K": 0.0, - "HumanEval": 0.0 - }, - "chatcore_metric": -0.011321608241655831, - "chatcore_suite": { - "name": "karpathy", - "tasks": [ - "ARC-Easy", - "ARC-Challenge", - "MMLU", - "GSM8K", - "HumanEval" - ], - "max_generative_problems": null, - "generative_answer_format": null - }, - "complete": true -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/run.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/run.json deleted file mode 100644 index f5bb2a33604ebc019414f11497210eca388f8274..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v2", - "stage": "sft", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_checkpoint_step": 9600, - "branch_parent_step": null, - "config_fingerprint": "4462a389dc98cd5d", - "wandb_run_id": "036b1c71", - "created_at": 1786715992 -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3-a100-40gb/config.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3-a100-40gb/config.json deleted file mode 100644 index a4ddf2ba6dc539aaf0dc31a8a6b692b20426b89a..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3-a100-40gb/config.json +++ /dev/null @@ -1,97 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c3-robust-v3-a100-40gb", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C3Rv3", - "mode": "staged", - "passes": 3, - "threshold_default": 80, - "stages": [ - { - "routes": [ - "knowledge_qa" - ], - "authentic": "single" - }, - { - "routes": [ - "reasoning_qa", - "stem_reasoning", - "how_to_qa", - "opinion_qa", - "composition_qa", - "verse_qa" - ], - "calibration_qa": true - }, - { - "routes": [ - "multiturn_qa", - "narrative_grounded", - "narrative_fiction" - ], - "authentic": "multi" - } - ], - "noise": { - "rate": 0.3, - "end_punct_rate": 0.05 - }, - "robustness": { - "stage": 0, - "epochs": 2.5, - "routes": { - "conversation_qa": { - "count": null - }, - "unparseable_qa": { - "count": null - }, - "typo_qa": { - "count": null - }, - "era_qa": { - "count": null - }, - "conversation_multiturn": { - "count": null - } - } - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 1, - "warmup_ratio": 0.03, - "eval_every": 400, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "curriculum", - "c3", - "scale-max", - "staged", - "robustness", - "noise", - "d32" - ] - }, - "config_fingerprint": "b0d39eef6f8e427b", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3-a100-40gb" -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3-a100-40gb/run.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3-a100-40gb/run.json deleted file mode 100644 index 7e26290a20db3dbd2a963181d46a4707e7763913..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3-a100-40gb/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3-a100-40gb", - "stage": "sft", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_checkpoint_step": 9600, - "branch_parent_step": null, - "config_fingerprint": "b0d39eef6f8e427b", - "wandb_run_id": "5667c0f5", - "created_at": 1787104196 -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3/config.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3/config.json deleted file mode 100644 index 8c521169515b02fe0eeb96ee888c1344007cb256..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3/config.json +++ /dev/null @@ -1,97 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c3-robust-v3", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C3Rv3", - "mode": "staged", - "passes": 3, - "threshold_default": 80, - "stages": [ - { - "routes": [ - "knowledge_qa" - ], - "authentic": "single" - }, - { - "routes": [ - "reasoning_qa", - "stem_reasoning", - "how_to_qa", - "opinion_qa", - "composition_qa", - "verse_qa" - ], - "calibration_qa": true - }, - { - "routes": [ - "multiturn_qa", - "narrative_grounded", - "narrative_fiction" - ], - "authentic": "multi" - } - ], - "noise": { - "rate": 0.3, - "end_punct_rate": 0.05 - }, - "robustness": { - "stage": 0, - "epochs": 2.5, - "routes": { - "conversation_qa": { - "count": null - }, - "unparseable_qa": { - "count": null - }, - "typo_qa": { - "count": null - }, - "era_qa": { - "count": null - }, - "conversation_multiturn": { - "count": null - } - } - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 2, - "warmup_ratio": 0.03, - "eval_every": 400, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "curriculum", - "c3", - "scale-max", - "staged", - "robustness", - "noise", - "d32" - ] - }, - "config_fingerprint": "12ffd193bdca635a", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3" -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3/run.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3/run.json deleted file mode 100644 index 49f8abf4bab7661292b549f3d45f760f967fdec0..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust-v3", - "stage": "sft", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_checkpoint_step": 9600, - "branch_parent_step": null, - "config_fingerprint": "12ffd193bdca635a", - "wandb_run_id": "10993832", - "created_at": 1787101675 -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/checkpoints/meta_000041.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/checkpoints/meta_000041.json deleted file mode 100644 index cf590f0bd077db8b95de142b78e85230c742e9ec..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/checkpoints/meta_000041.json +++ /dev/null @@ -1,172 +0,0 @@ -{ - "step": 41, - "training_complete": true, - "val_bpb": 0.7046443464143689, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust", - "wandb_run_id": "21793571", - "wandb_group": "think-d32", - "wandb_tags": "sft,curriculum,c3,scale-max,staged,robustness,noise,d32", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/base_checkpoints", - "base_step": 9600, - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/checkpoints", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer", - "resume_from_step": null, - "experiment_id": "Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/config.json", - "parent_cumulative_flops": 2.3103503196928082e+20, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "8549a37b028418b0abaf347bd6e5a8ccb6cff782", - "load_optimizer": 0, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 2, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.03, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": 400, - "eval_tokens": 20971520, - "chatcore_every": 100000, - "chatcore_max_cat": -1, - "chatcore_max_sample": 32, - "save_every": -1, - "recipe": "curriculum", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c3-robust", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C3R", - "mode": "staged", - "threshold_default": 80, - "stages": [ - { - "routes": [ - "knowledge_qa" - ], - "authentic": "single" - }, - { - "routes": [ - "reasoning_qa", - "stem_reasoning", - "how_to_qa", - "opinion_qa", - "composition_qa", - "verse_qa" - ], - "calibration_qa": true - }, - { - "routes": [ - "multiturn_qa", - "narrative_grounded", - "narrative_fiction" - ], - "authentic": "multi" - } - ], - "noise": { - "rate": 0.3 - }, - "robustness": { - "stage": 0, - "epochs": 1, - "routes": { - "conversation_qa": { - "count": null - }, - "unparseable_qa": { - "count": null - }, - "era_qa": { - "count": null - } - } - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 2, - "warmup_ratio": 0.03, - "eval_every": 400, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "curriculum", - "c3", - "scale-max", - "staged", - "robustness", - "noise", - "d32" - ] - }, - "config_fingerprint": "e1a9d25fb1cf5edf", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "e1a9d25fb1cf5edf" - }, - "loop_state": { - "step": 41, - "total_training_time": 1459.9074046611786, - "min_val_bpb": 0.7046443464143689, - "smooth_train_loss": 1.907550057536094, - "mfu": 51.698687069018604, - "tok_per_sec": 44555, - "stage_training_flops": 9.867121157021368e+17, - "inherited_parent_flops": 2.3103503196928082e+20, - "cumulative_pipeline_training_flops": 2.3202174408498296e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/checkpoints/model_000041.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/checkpoints/model_000041.pt deleted file mode 100644 index 99714f4aba77f97c5899d2f390de263ff59385f4..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/checkpoints/model_000041.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:590b64a7ab561ae7ec7adf10fd09469e7b46dacf76e5c665f743fd9d0dd74a6f -size 8992694653 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/checkpoints/optim_000041_rank0.pt b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/checkpoints/optim_000041_rank0.pt deleted file mode 100644 index 16a2b5c288bb24620b6506caef0127d5e405cce0..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/checkpoints/optim_000041_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7c642c3e04109fce332ca03b7afaf571812e0d9ac124f8863889f262a6748995 -size 11545903753 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/config.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/config.json deleted file mode 100644 index 489161da052f062973e9fa79d3a4ed90fc733957..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/config.json +++ /dev/null @@ -1,89 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c3-robust", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C3R", - "mode": "staged", - "threshold_default": 80, - "stages": [ - { - "routes": [ - "knowledge_qa" - ], - "authentic": "single" - }, - { - "routes": [ - "reasoning_qa", - "stem_reasoning", - "how_to_qa", - "opinion_qa", - "composition_qa", - "verse_qa" - ], - "calibration_qa": true - }, - { - "routes": [ - "multiturn_qa", - "narrative_grounded", - "narrative_fiction" - ], - "authentic": "multi" - } - ], - "noise": { - "rate": 0.3 - }, - "robustness": { - "stage": 0, - "epochs": 1, - "routes": { - "conversation_qa": { - "count": null - }, - "unparseable_qa": { - "count": null - }, - "era_qa": { - "count": null - } - } - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 2, - "warmup_ratio": 0.03, - "eval_every": 400, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d32", - "tags": [ - "sft", - "curriculum", - "c3", - "scale-max", - "staged", - "robustness", - "noise", - "d32" - ] - }, - "config_fingerprint": "e1a9d25fb1cf5edf", - "artifact_path": "experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust" -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/run.json b/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/run.json deleted file mode 100644 index 702fe685f0d2c9a3a50218e829b938ac8b8e4032..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/sft/Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont-pre1930-curriculum-c3-robust", - "stage": "sft", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_checkpoint_step": 9600, - "branch_parent_step": null, - "config_fingerprint": "e1a9d25fb1cf5edf", - "wandb_run_id": "21793571", - "created_at": 1786635058 -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/summary.json b/experiments/Think.Unbounded-d32-v2mix-cont/summary.json deleted file mode 100644 index bab0e720b7ef299b29a99cb887fa99992d12b983..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/summary.json +++ /dev/null @@ -1,79 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32-v2mix-cont", - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32-v2mix-cont", - "parent_experiment_id": "Think.Unbounded-d32", - "parent_checkpoint_step": 5500, - "dataset": "jbduran/think-dataset-clean-1930s", - "dataset_revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "step": 9600, - "depth": 32, - "target_param_data_ratio": 12, - "training_tokens": 20132659200, - "final_sampled_val_bpb": 0.7708812390249374, - "minimum_sampled_val_bpb": 0.7701763237939314, - "full_val_bpb": 0.7505380497637221, - "core_metric": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the most important city in the world. It is the capital of the most powerful" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is Au, and that of silver is Ag. The symbol of the former is Ag" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If yesterday was Saturday, then tomorrow will be Sunday. If yesterday was" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is cold, and the opposite of cold is hot. The opposite of hot is cold" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: Mercury, Venus, Mars, Jupiter, Saturn, and Herschel. The earth" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a deep, rich, velvety crimson, the same shade as the rose" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is 5.\n\nIf 3x + 4 = 13, then x" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nTHIS essay originated with Mr. W. C. Barrett, who says of it: \"Mr. Galton's 'Life and Times,' and his Life of Darwin.' \"It is mainly a half-tone reprint from Galton's (vice supra) life, to which I have added some important facts which I had not been able to procure when the original edition was published, and which bear directly upon some matters of interest.\n\nIt may, however, be as well to draw attention more particularly to the point to which Mr. Barrett refers in the above quotation, as this contains, I believe, the fundamental distinction between the Darwin", - "<|bos|>Lady Saumarez -- Scotch rose shaded white. Pheasant's Eye -- Pure white, dark eye. Marigold -- Striking light golden color. Maddereti -- A most desirable purple.\n\nRose -- -A superb full flowering rose of a most beautiful shade of deep rose. Sunrise -- Pure tints of fawn and yellow.\n\nCarnation Marguerite\n\nLarge Flowered\n\nnow offered. Much hardier and blooms more abundantly.\n\nPrice, any variety, 10 cts. each; set of 1 3 for 45 cents.\n\nEasily Makes a Good Center\n\nAnywhere.\n\n", - "<|bos|>RIDGEN, TOLLAND\n\nFRIDLAND, Dyfe, y 1665.\n\nTHE CONFUCTIONS\n\nor\n\nTHE KINGDOM OF SOLOMON.\n\nTHE FIRST BOOK OF THE KING OF SOLOMON\n\nRefers to the Four Series of the Four Dispensations, God revealing himself to His Children in the various Births of His Israel. The Messiah's birth and baptism of John, the mission of the\n\nApostles, and the baptism of Christ, and latterly the majestic progress of His power, are chronicled in the Books of Moses, of the Prophets, of the Proph ets, and the", - "<|bos|>agers have already told us, and M. Crookes has already informed us, that he found concretions of this nature in the plantlice when giving a supplementary examination to them : hence, even according to his own showing, no safe conclusion can be drawn from the relative order and distribution of these silica-precipitates. At the same time it is natural to ask whether this tardy means of elucidation, instead of becoming antiquated by the lapse of the fair period which Hibbert has now completed-and well-nigh brought to a close by a misuse of some words, may not be pressed into use - --\n\nON", - "<|bos|>D\n\nio\n\nLe\n\nthered in the middle sept i post- ater bear ried ne pro avera hing tenanted\n\nCrowne he se ouch h find\n\n\u200bLa thi seeks depone \u0f51 \u0f4f \u0f63\u0f74 \u0f49\u0f7a\u0f51 \u0f61 \u0f61 \u0f61\u0f72 \u0f64\u0f74 \u0f62 \u0f61 \u0f51 \u0f62 \u3002 \u0f61\u0f72\u0970\u0f62 \u0f51 \u0f0b \u0f62\u0f66 Atth Id\n\nDedication of the Picture in the Library of Trinity that has the most frequently called forth criticism\n\nI am unable to", - "<|bos|>PREFACE\n\nCONTENTS.\n\n1. X. About a month was being wasted at the hotel, when a kindhearted friend offered to go on ahead and engage rooms for us at the best houses the day was hot-the Southern hotels always are-on the other side, and on most of them, indeed, fair profit can be made; and the diligence kept always waiting at the station and often not in readiness for an hour after the last customer had been put in, and the engineer grumbled fearfully at this irregularity of the train, because the trains gave him a pretty lively time when there was anybody waiting there to see, and the porters sometimes began", - "<|bos|>Clarke Model\n\n222000908000 DIANER DER\n\nThe proper unture dishes\n\nWilder Damases7", - "<|bos|>ossoer 0 020 oO 000000 ee OP atte eT es ee eee en eee Bee Se SERRE Nee errre reser Res eererr ee eee cerr reeer era Nee essen re ers 2008 OO OO DO ES OO ESE HD TTI MT SPEC IC PRT APOLOC H OCDOETTES 50 Ce a OC ee DDg ee CF EIOLUS) oe eA tt } 500 een eeehey n" - ], - "training_time_seconds": 164042.2579112053, - "stage_training_flops": 9.867121157021368e+19, - "inherited_parent_flops": 1.3236382039906714e+20, - "cumulative_pipeline_training_flops": 2.3103503196928082e+20, - "config_fingerprint": "00af0bde578c1269", - "git_commit_sha": "734396fce10c8fcd7bf979261d41ebe37968089c", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/2fa1b744", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/Think.Unbounded-d32-v2mix-cont", - "dataset_fingerprint": "54cc47b13796e411", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "unique_train_tokens_per_source": { - "original": 14515647284, - "midtrain_r21": 4107062477, - "midtrain_r45": 2053531239 - }, - "branch_parent_experiment_id": "Think.Unbounded-d32", - "branch_parent_step": 5500, - "branch_lr_schedule": "continue", - "stage_training_tokens": 8598323200, - "unique_train_tokens": 20676241000, - "effective_epochs": 0.41585524177242855 -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer/experiment_tokenizer.json b/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer/token_bytes.pt b/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer/tokenizer.pkl b/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32-v2mix-cont/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/meta_000500.json b/experiments/Think.Unbounded-d32/base_checkpoints/meta_000500.json deleted file mode 100644 index 7935dcc7dfb651299c72419dd22853c0209a9570..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,236 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.1476923999299866, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32", - "wandb_run_id": "9b75691b", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_original\", \"midtrain_r30\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "experiment_id": "Think.Unbounded-d32", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "b2de514e00cde3bee362cf4c87e16d35e0f32c24", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 391, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 207, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "6f8ae461e0eb9ee2", - "artifact_path": "experiments/Think.Unbounded-d32" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6f8ae461e0eb9ee2" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48712193, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48712193, - "mixture": { - "cursors": { - "original": { - "file_idx": 10, - "pos": 48712193, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48712193 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 1048584192, - "source_tokens": { - "original": 1048584192, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 1.1476923999299866, - "smooth_train_loss": 3.427469208772481, - "total_training_time": 19713.592538118362, - "stage_start_step": 0, - "stage_training_flops": 12033074581733376000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.2033074581733376e+19 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/meta_001000.json b/experiments/Think.Unbounded-d32/base_checkpoints/meta_001000.json deleted file mode 100644 index 219c4ab8436d17235fa15e955c056b199bad077e..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,236 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.056581441012716, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32", - "wandb_run_id": "9b75691b", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_original\", \"midtrain_r30\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "experiment_id": "Think.Unbounded-d32", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "b2de514e00cde3bee362cf4c87e16d35e0f32c24", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 391, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 207, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "6f8ae461e0eb9ee2", - "artifact_path": "experiments/Think.Unbounded-d32" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6f8ae461e0eb9ee2" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 20, - "pos": 97416193, - "epoch": 1, - "pq_idx": 20, - "rg_idx": 97416193, - "mixture": { - "cursors": { - "original": { - "file_idx": 20, - "pos": 97416193, - "epoch": 1, - "pq_idx": 20, - "rg_idx": 97416193 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 2097160192, - "source_tokens": { - "original": 2097160192, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 1.056581441012716, - "smooth_train_loss": 2.9296221567382985, - "total_training_time": 39794.540657281876, - "stage_start_step": 0, - "stage_training_flops": 24066149163466752000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2.406614916346675e+19 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/meta_001500.json b/experiments/Think.Unbounded-d32/base_checkpoints/meta_001500.json deleted file mode 100644 index 0f1ef0a6c9588eacb2fa8c3569d9d4a85ddad695..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,236 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.0152464516907762, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32", - "wandb_run_id": "9b75691b", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_original\", \"midtrain_r30\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "experiment_id": "Think.Unbounded-d32", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "b2de514e00cde3bee362cf4c87e16d35e0f32c24", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 391, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 207, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "6f8ae461e0eb9ee2", - "artifact_path": "experiments/Think.Unbounded-d32" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6f8ae461e0eb9ee2" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 31, - "pos": 46120193, - "epoch": 1, - "pq_idx": 31, - "rg_idx": 46120193, - "mixture": { - "cursors": { - "original": { - "file_idx": 31, - "pos": 46120193, - "epoch": 1, - "pq_idx": 31, - "rg_idx": 46120193 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 3145736192, - "source_tokens": { - "original": 3145736192, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 1.0152464516907762, - "smooth_train_loss": 2.935989382480182, - "total_training_time": 59860.37234377861, - "stage_start_step": 0, - "stage_training_flops": 36099223745200128000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 3.609922374520013e+19 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/meta_002000.json b/experiments/Think.Unbounded-d32/base_checkpoints/meta_002000.json deleted file mode 100644 index f2a7acb0803feb91d39817fcd45a4a04b41b5375..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,236 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 0.9928518912323213, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32", - "wandb_run_id": "9b75691b", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_original\", \"midtrain_r30\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "experiment_id": "Think.Unbounded-d32", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "b2de514e00cde3bee362cf4c87e16d35e0f32c24", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 391, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 207, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "6f8ae461e0eb9ee2", - "artifact_path": "experiments/Think.Unbounded-d32" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6f8ae461e0eb9ee2" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 41, - "pos": 94824193, - "epoch": 1, - "pq_idx": 41, - "rg_idx": 94824193, - "mixture": { - "cursors": { - "original": { - "file_idx": 41, - "pos": 94824193, - "epoch": 1, - "pq_idx": 41, - "rg_idx": 94824193 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 4194312192, - "source_tokens": { - "original": 4194312192, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 0.9928518912323213, - "smooth_train_loss": 2.715263738468478, - "total_training_time": 79915.32717370987, - "stage_start_step": 0, - "stage_training_flops": 48132298326933504000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 4.81322983269335e+19 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/meta_002500.json b/experiments/Think.Unbounded-d32/base_checkpoints/meta_002500.json deleted file mode 100644 index fc94ea366818b97bfde1f6f3cf5da6c55565bdb2..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/meta_002500.json +++ /dev/null @@ -1,236 +0,0 @@ -{ - "step": 2500, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 0.9737910804428661, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32", - "wandb_run_id": "9b75691b", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_original\", \"midtrain_r30\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "experiment_id": "Think.Unbounded-d32", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "b2de514e00cde3bee362cf4c87e16d35e0f32c24", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 391, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 207, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "6f8ae461e0eb9ee2", - "artifact_path": "experiments/Think.Unbounded-d32" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6f8ae461e0eb9ee2" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 52, - "pos": 43528193, - "epoch": 1, - "pq_idx": 52, - "rg_idx": 43528193, - "mixture": { - "cursors": { - "original": { - "file_idx": 52, - "pos": 43528193, - "epoch": 1, - "pq_idx": 52, - "rg_idx": 43528193 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 5242888192, - "source_tokens": { - "original": 5242888192, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 0.9630578341851044, - "smooth_train_loss": 2.7360114404103504, - "total_training_time": 99964.06782126427, - "stage_start_step": 0, - "stage_training_flops": 60165372908666880000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 6.016537290866688e+19 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/meta_003000.json b/experiments/Think.Unbounded-d32/base_checkpoints/meta_003000.json deleted file mode 100644 index b8a81d5b057170634b33245f418ef0a6a73a246e..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/meta_003000.json +++ /dev/null @@ -1,236 +0,0 @@ -{ - "step": 3000, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 0.9735215125930343, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32", - "wandb_run_id": "9b75691b", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_original\", \"midtrain_r30\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "experiment_id": "Think.Unbounded-d32", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "b2de514e00cde3bee362cf4c87e16d35e0f32c24", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 391, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 207, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "6f8ae461e0eb9ee2", - "artifact_path": "experiments/Think.Unbounded-d32" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6f8ae461e0eb9ee2" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 62, - "pos": 92232193, - "epoch": 1, - "pq_idx": 62, - "rg_idx": 92232193, - "mixture": { - "cursors": { - "original": { - "file_idx": 62, - "pos": 92232193, - "epoch": 1, - "pq_idx": 62, - "rg_idx": 92232193 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 6291464192, - "source_tokens": { - "original": 6291464192, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 0.9630578341851044, - "smooth_train_loss": 2.6967057059495345, - "total_training_time": 120021.50824666023, - "stage_start_step": 0, - "stage_training_flops": 72198447490400256000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 7.219844749040026e+19 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/meta_003500.json b/experiments/Think.Unbounded-d32/base_checkpoints/meta_003500.json deleted file mode 100644 index 8e770ca7582b046b206cc7188a4512c19d2b6b6a..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/meta_003500.json +++ /dev/null @@ -1,236 +0,0 @@ -{ - "step": 3500, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 0.968207957699399, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32", - "wandb_run_id": "9b75691b", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_original\", \"midtrain_r30\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "experiment_id": "Think.Unbounded-d32", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "b2de514e00cde3bee362cf4c87e16d35e0f32c24", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 391, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 207, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "6f8ae461e0eb9ee2", - "artifact_path": "experiments/Think.Unbounded-d32" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6f8ae461e0eb9ee2" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 73, - "pos": 40936193, - "epoch": 1, - "pq_idx": 73, - "rg_idx": 40936193, - "mixture": { - "cursors": { - "original": { - "file_idx": 73, - "pos": 40936193, - "epoch": 1, - "pq_idx": 73, - "rg_idx": 40936193 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 7340040192, - "source_tokens": { - "original": 7340040192, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 0.9630578341851044, - "smooth_train_loss": 2.5911028486222323, - "total_training_time": 140049.3800020218, - "stage_start_step": 0, - "stage_training_flops": 84231522072133632000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 8.423152207213363e+19 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/meta_004000.json b/experiments/Think.Unbounded-d32/base_checkpoints/meta_004000.json deleted file mode 100644 index 29c848cb860c77857b9d61e09ca9036b75102b08..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/meta_004000.json +++ /dev/null @@ -1,236 +0,0 @@ -{ - "step": 4000, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 0.9585141600611977, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32", - "wandb_run_id": "9b75691b", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_original\", \"midtrain_r30\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "experiment_id": "Think.Unbounded-d32", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "b2de514e00cde3bee362cf4c87e16d35e0f32c24", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 391, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 207, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "6f8ae461e0eb9ee2", - "artifact_path": "experiments/Think.Unbounded-d32" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6f8ae461e0eb9ee2" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 83, - "pos": 89640193, - "epoch": 1, - "pq_idx": 83, - "rg_idx": 89640193, - "mixture": { - "cursors": { - "original": { - "file_idx": 83, - "pos": 89640193, - "epoch": 1, - "pq_idx": 83, - "rg_idx": 89640193 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 8388616192, - "source_tokens": { - "original": 8388616192, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 0.9585141600611977, - "smooth_train_loss": 2.7874954028564223, - "total_training_time": 160072.53116488457, - "stage_start_step": 0, - "stage_training_flops": 96264596653867008000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 9.6264596653867e+19 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/meta_004500.json b/experiments/Think.Unbounded-d32/base_checkpoints/meta_004500.json deleted file mode 100644 index 17dc68edc7e4507427eb4d45adb1098c37e42880..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/meta_004500.json +++ /dev/null @@ -1,236 +0,0 @@ -{ - "step": 4500, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 0.956614818178498, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32", - "wandb_run_id": "9b75691b", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_original\", \"midtrain_r30\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "experiment_id": "Think.Unbounded-d32", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "b2de514e00cde3bee362cf4c87e16d35e0f32c24", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 391, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 207, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "6f8ae461e0eb9ee2", - "artifact_path": "experiments/Think.Unbounded-d32" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6f8ae461e0eb9ee2" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 94, - "pos": 38344193, - "epoch": 1, - "pq_idx": 94, - "rg_idx": 38344193, - "mixture": { - "cursors": { - "original": { - "file_idx": 94, - "pos": 38344193, - "epoch": 1, - "pq_idx": 94, - "rg_idx": 38344193 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 9437192192, - "source_tokens": { - "original": 9437192192, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 0.956614818178498, - "smooth_train_loss": 2.591393936107061, - "total_training_time": 180101.40045905113, - "stage_start_step": 0, - "stage_training_flops": 108297671235600384000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.0829767123560038e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/meta_005000.json b/experiments/Think.Unbounded-d32/base_checkpoints/meta_005000.json deleted file mode 100644 index 09794af0eec806aff048c4816ae11da7f4fdc682..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/meta_005000.json +++ /dev/null @@ -1,236 +0,0 @@ -{ - "step": 5000, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 0.9363505794880082, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32", - "wandb_run_id": "9b75691b", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_original\", \"midtrain_r30\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "experiment_id": "Think.Unbounded-d32", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "b2de514e00cde3bee362cf4c87e16d35e0f32c24", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 391, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 207, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "6f8ae461e0eb9ee2", - "artifact_path": "experiments/Think.Unbounded-d32" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6f8ae461e0eb9ee2" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 104, - "pos": 87048193, - "epoch": 1, - "pq_idx": 104, - "rg_idx": 87048193, - "mixture": { - "cursors": { - "original": { - "file_idx": 104, - "pos": 87048193, - "epoch": 1, - "pq_idx": 104, - "rg_idx": 87048193 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 10485768192, - "source_tokens": { - "original": 10485768192, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 0.9363505794880082, - "smooth_train_loss": 2.4582942112751494, - "total_training_time": 200133.8836593628, - "stage_start_step": 0, - "stage_training_flops": 120330745817333760000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.2033074581733376e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/meta_005500.json b/experiments/Think.Unbounded-d32/base_checkpoints/meta_005500.json deleted file mode 100644 index 2e539305014a734dedb0efd28cc56e73a826945e..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/meta_005500.json +++ /dev/null @@ -1,236 +0,0 @@ -{ - "step": 5500, - "training_complete": false, - "experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 0.8999200969809518, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 32, - "n_head": 16, - "n_kv_head": 16, - "n_embd": 2048, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "Think.Unbounded-d32", - "wandb_run_id": "9b75691b", - "wandb_group": "clean1930s-d32", - "wandb_tags": "think-dataset-clean-1930s,d32,ratio12,ctx4096,sssl,fp8,muon-momentum-constant,muon-momentum-0.90,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 32, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 9600, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 2, - "total_batch_size": 2097152, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/data", - "tokenizer_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/pretok", - "mixture_source_dirs": "{\"original\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_original\", \"midtrain_r30\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/Think.Unbounded-d32/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/Think.Unbounded-d32/base_checkpoints", - "experiment_id": "Think.Unbounded-d32", - "experiment_config": "/workspace/nanochat/experiments/Think.Unbounded-d32/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "b2de514e00cde3bee362cf4c87e16d35e0f32c24", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "Think.Unbounded-d32", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 391, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 207, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "6f8ae461e0eb9ee2", - "artifact_path": "experiments/Think.Unbounded-d32" - }, - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6f8ae461e0eb9ee2" - }, - "device_batch_size": 2, - "max_seq_len": 4096, - "total_batch_size": 2097152, - "dataloader_state_dict": { - "file_idx": 115, - "pos": 35752193, - "epoch": 1, - "pq_idx": 115, - "rg_idx": 35752193, - "mixture": { - "cursors": { - "original": { - "file_idx": 115, - "pos": 35752193, - "epoch": 1, - "pq_idx": 115, - "rg_idx": 35752193 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 11534344192, - "source_tokens": { - "original": 11534344192, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 0.8999200969809518, - "smooth_train_loss": 2.4195481055384045, - "total_training_time": 220148.93892598152, - "stage_start_step": 0, - "stage_training_flops": 132363820399067136000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.3236382039906714e+20 - } -} \ No newline at end of file diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/model_000500.pt b/experiments/Think.Unbounded-d32/base_checkpoints/model_000500.pt deleted file mode 100644 index 6884c1da4747df1ea45e3a188c830568c8218441..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:adc847ba9a4b6768498bbf6b84212825ac86098a137a340fd2eaefa72c01a021 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/model_001000.pt b/experiments/Think.Unbounded-d32/base_checkpoints/model_001000.pt deleted file mode 100644 index 6edd271c451f5fe040f5cfda073887ec81dacf16..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2ec635fc9fd654db070738b9ebc8b84f8b1169d60682241c8836cde1b5bf8a71 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/model_001500.pt b/experiments/Think.Unbounded-d32/base_checkpoints/model_001500.pt deleted file mode 100644 index 07f84dcfd96105cd888bf167f91d56ebd9f7b0d1..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2da68abcf5dbd78f3af6b00b815a538bda1955d7977ba49697b0aec3843bab2f -size 8992694653 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/model_002000.pt b/experiments/Think.Unbounded-d32/base_checkpoints/model_002000.pt deleted file mode 100644 index 4afc57a233c2c1ad2aa774b1be488ea4978f19a7..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d7e811bb7d332aa76a8593587752f27efcea20af3a03a221d2ae3a7abad4aef2 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/model_002500.pt b/experiments/Think.Unbounded-d32/base_checkpoints/model_002500.pt deleted file mode 100644 index e87141a787975437d3ebfdbf04972e8d1492e458..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e9bc3c0f1ed3f5510185c16226614281cc7187b71ddcf53447054985be764722 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/model_003000.pt b/experiments/Think.Unbounded-d32/base_checkpoints/model_003000.pt deleted file mode 100644 index 4b1a13fc0d996c94bbadf6efbdbb7eb3d3700d61..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/model_003000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e6f2485fd307a258eb3fded1bc9b1be934ac2184fbf2eefdf10d850bba36cb13 -size 8992694653 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/model_003500.pt b/experiments/Think.Unbounded-d32/base_checkpoints/model_003500.pt deleted file mode 100644 index fdd15528b38ba1332fb8ca09e5da6525423ea247..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/model_003500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:cc4d43cb19a8d1c085f0d3a9ee21b023ea1ee5a4f573896f628f2f169503aeaa -size 8992694653 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/model_004000.pt b/experiments/Think.Unbounded-d32/base_checkpoints/model_004000.pt deleted file mode 100644 index 6cea76a8152563913ca242d8c2c6a6606f2f2401..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/model_004000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d1c498d9afbfe6ef3231f42221c888f3cf89e17ed6526b59562b40b1df47475a -size 8992694653 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/model_004500.pt b/experiments/Think.Unbounded-d32/base_checkpoints/model_004500.pt deleted file mode 100644 index 7cc16fe8f66c528b12b60c607f1985a0cad2cc63..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/model_004500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9bba2104bb0b0dc658dc52e9cc4c31ec0356e0d051f30e8ddb3d1dce564008ce -size 8992694653 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/model_005000.pt b/experiments/Think.Unbounded-d32/base_checkpoints/model_005000.pt deleted file mode 100644 index c45be74bd3c00d3a7a8c840d1822607f3d767825..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/model_005000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7603f520faeabb51df975a35ab8b0065e5ec14782eceb8077dca635045d05f1a -size 8992694653 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/model_005500.pt b/experiments/Think.Unbounded-d32/base_checkpoints/model_005500.pt deleted file mode 100644 index 59d5d22a28f4b6c92f97d4294863afe521af77a3..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/model_005500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e40a97ef3206dbb1c16ebcb1d587055eaee6438f057d150ba2f0df78345b372a -size 8992694653 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/optim_000500_rank0.pt b/experiments/Think.Unbounded-d32/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 23ee1b6e135c15acdb5fc9bd73271bfadd44d8fe..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:060bd64c5f41a991f7822484d4856efc90683c9530eed9fb3ffbdf286428415f -size 11545903817 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/optim_001000_rank0.pt b/experiments/Think.Unbounded-d32/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 925866ef2023e9cd594d3ce1896f2cba6dedb74d..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b93c1121cdb0fb1a70add6e4ce189ea7d2fc79ec6b0348ecddcc5b3e8e141ac4 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/optim_001500_rank0.pt b/experiments/Think.Unbounded-d32/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index cdde4199fd7933bce9164408cfbc97c3635b275e..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:18141073e073d931961e0c105037de5db7e46113e84f0bfd9faac16e92a8dd6e -size 11545903817 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/optim_002000_rank0.pt b/experiments/Think.Unbounded-d32/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index df33ef462ed8be0f02fc0a063a55a6333dc55fbd..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fb065a47e6c313adf94f475ef87686934efc532c0e0708aa19ab1fb18778074d -size 11545903817 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/optim_002500_rank0.pt b/experiments/Think.Unbounded-d32/base_checkpoints/optim_002500_rank0.pt deleted file mode 100644 index 7938fc4fcfa6127a18b84d327c9bea9a7c299955..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b23c21030cd25dc859e938236993159aa0c951398c99904083e984cd8c9208f5 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/optim_003000_rank0.pt b/experiments/Think.Unbounded-d32/base_checkpoints/optim_003000_rank0.pt deleted file mode 100644 index 1654df1b6cf3d91ad0dacefcb30ad11790aa4907..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/optim_003000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e56c1e1a3a6e4eff55d3df7eb840e1a2ed60b9e314c1a854ccdbcf2f2a498aee -size 11545903817 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/optim_003500_rank0.pt b/experiments/Think.Unbounded-d32/base_checkpoints/optim_003500_rank0.pt deleted file mode 100644 index 81898c64322dc9dba9588808061c557fdd365637..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/optim_003500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:09542bfde4246e07a20f2f250562204f2b03c8044963b00cdb30808c357e8f94 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/optim_004000_rank0.pt b/experiments/Think.Unbounded-d32/base_checkpoints/optim_004000_rank0.pt deleted file mode 100644 index 50722b5b393b9445ebe90e25204add3f2673ddee..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/optim_004000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c0880f6566493f92ee6c594d3c59eb79c839931565a44014b8cde39340196901 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/optim_004500_rank0.pt b/experiments/Think.Unbounded-d32/base_checkpoints/optim_004500_rank0.pt deleted file mode 100644 index 729b852acebc026e69c0bd66b96df39230026402..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/optim_004500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:045613067d60df6f12928ed7b71754c30712819a1892ecb1507b8d4786b4fd52 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/optim_005000_rank0.pt b/experiments/Think.Unbounded-d32/base_checkpoints/optim_005000_rank0.pt deleted file mode 100644 index eeaacecf2c1b9dd5a333ba4f64a42e83f980cd1a..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/optim_005000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4bd4ccad84654c8e38232615469d44d622fff0348893ce28a7613db4a67845f6 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32/base_checkpoints/optim_005500_rank0.pt b/experiments/Think.Unbounded-d32/base_checkpoints/optim_005500_rank0.pt deleted file mode 100644 index 603391a7f5229a41de7b825493fb3dc2e8f95054..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/base_checkpoints/optim_005500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6fcf19df437fd1aa1f97c344b072ff93f6254e84a4733cd6af25969385faeb39 -size 11545903817 diff --git a/experiments/Think.Unbounded-d32/config.json b/experiments/Think.Unbounded-d32/config.json deleted file mode 100644 index b56575dc5df745833e8b4c9d3fbc593d69ce3359..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/config.json +++ /dev/null @@ -1,111 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "Think.Unbounded-d32", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "45225d95bc15f942be3b4b344738cea1f66e3de8", - "validation_shard": 472, - "num_train_shards": 330, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 391, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "9ace24b8e16e57e38a1ea0b1f6d7cbf323bf9dbc", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 207, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 20132659200, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 14092861440, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 18119393280, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 32, - "scaling_params": 1677724672, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 2, - "total_batch_size": 2097152, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "Think.Unbounded-d32", - "group": "clean1930s-d32", - "tags": [ - "think-dataset-clean-1930s", - "d32", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "muon-momentum-constant", - "muon-momentum-0.90", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "6f8ae461e0eb9ee2", - "artifact_path": "experiments/Think.Unbounded-d32" -} diff --git a/experiments/Think.Unbounded-d32/run.json b/experiments/Think.Unbounded-d32/run.json deleted file mode 100644 index 6826cd812486d84f4a4dc7e69abc8bd1e2194015..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "Think.Unbounded-d32", - "stage": "base", - "base_experiment_id": "Think.Unbounded-d32", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "branch_parent_step": null, - "config_fingerprint": "6f8ae461e0eb9ee2", - "wandb_run_id": "9b75691b", - "created_at": 1785990696 -} diff --git a/experiments/Think.Unbounded-d32/tokenizer/experiment_tokenizer.json b/experiments/Think.Unbounded-d32/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/Think.Unbounded-d32/tokenizer/token_bytes.pt b/experiments/Think.Unbounded-d32/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/Think.Unbounded-d32/tokenizer/tokenizer.pkl b/experiments/Think.Unbounded-d32/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/Think.Unbounded-d32/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_000500.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_000500.json deleted file mode 100644 index e41094835a7e09f53fd78b890bad5c730d50d873..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,147 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "val_bpb": 1.341851696959099, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "wandb_run_id": "73b2700b", - "wandb_group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,attention-ablation-v1,full-attention,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "attention-ablation-v1", - "full-attention", - "fp8" - ] - }, - "config_fingerprint": "9d3bb595965c289e", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "9d3bb595965c289e" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.341851696959099, - "smooth_train_loss": 3.7025805680376496, - "total_training_time": 571.568044424057, - "stage_training_flops": 291921016651776000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 291921016651776000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_001000.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_001000.json deleted file mode 100644 index 7e937e5bd51f62ee21bb6d658eb891b1b7b6f686..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,147 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "val_bpb": 1.2316896297206081, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "wandb_run_id": "73b2700b", - "wandb_group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,attention-ablation-v1,full-attention,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "attention-ablation-v1", - "full-attention", - "fp8" - ] - }, - "config_fingerprint": "9d3bb595965c289e", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "9d3bb595965c289e" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2316896297206081, - "smooth_train_loss": 3.515503448119396, - "total_training_time": 1155.6556208133698, - "stage_training_flops": 583842033303552000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 583842033303552000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_001500.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_001500.json deleted file mode 100644 index bb1f330e9d30d34481340516d19a96df77a19914..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,147 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "val_bpb": 1.14883061633963, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "wandb_run_id": "73b2700b", - "wandb_group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,attention-ablation-v1,full-attention,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "attention-ablation-v1", - "full-attention", - "fp8" - ] - }, - "config_fingerprint": "9d3bb595965c289e", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "9d3bb595965c289e" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.14883061633963, - "smooth_train_loss": 3.3998494498727863, - "total_training_time": 1739.6899712085724, - "stage_training_flops": 875763049955328000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 875763049955328000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_002000.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_002000.json deleted file mode 100644 index 246080de6b5fc40cb0e09a8a6819f263a332b20c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,147 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "val_bpb": 1.1118656088331524, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "wandb_run_id": "73b2700b", - "wandb_group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,attention-ablation-v1,full-attention,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "attention-ablation-v1", - "full-attention", - "fp8" - ] - }, - "config_fingerprint": "9d3bb595965c289e", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "9d3bb595965c289e" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.1118656088331524, - "smooth_train_loss": 3.330046126752348, - "total_training_time": 2324.030104160309, - "stage_training_flops": 1167684066607104000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1167684066607104000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_002362.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_002362.json deleted file mode 100644 index 735e8ebd7601898f6f317a1e2858d9149903cb7a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,147 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "val_bpb": 1.089810417727749, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "wandb_run_id": "73b2700b", - "wandb_group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,attention-ablation-v1,full-attention,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "attention-ablation-v1", - "full-attention", - "fp8" - ] - }, - "config_fingerprint": "9d3bb595965c289e", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "9d3bb595965c289e" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": 1.089810417727749, - "smooth_train_loss": 3.1332932870678034, - "total_training_time": 2747.30322766304, - "stage_training_flops": 1379034882662989824, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1379034882662989824 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_000500.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_000500.pt deleted file mode 100644 index dcb4351b6ec1cf82ae77bd2537c2852d2e7ef264..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:df0ad4d77d44ef41b61e65b9f2a2ef7ee6440fab7d14811fee9a3fcdc9e64e8e -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_001000.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_001000.pt deleted file mode 100644 index 8380f15ca9ab12b6d447e5a45824c9a482b4c33e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b2fae8d3f799e0de7533c4f89f72c8e706d7a4bce41f4b3f78b189687e532459 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_001500.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_001500.pt deleted file mode 100644 index 583ef04f8f412c28e802f5c7cc7125cc148f2567..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:53e1ecaed8685b05ba89e76679083569bc64a982e466051e1e41dfce8cb5dd63 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_002000.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_002000.pt deleted file mode 100644 index 7b15ab3979e242fba3c1561ebabc9d264fc4e52d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6cb70f5a7eced55f0298fab3a221652724d9292db9af53ab6ce2378f0a688439 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_002362.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_002362.pt deleted file mode 100644 index b969bf1f0dac779213d4e932ceb71e69f7bfb4b5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3ecc02a89c79321d39c942b4849b2e4d8a2e7c44b74d478a9b7457cb32a8eafa -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_000500_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 0b11bcb7eb1c34fb82fbe5478c88f1180ea6a02a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b261a4019a5d0f54a3b8b4a1f0677caeb489870e7e22107d7475a434f94c2ee0 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_001000_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index c9779371dd1e58446a431fa5d14169f7abd86129..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:83c213a1fb6ee27e243741c0337ffe1c60a659e93e44a8dd7307c7636a16dbec -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_001500_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 4f6ced2ad593c659ac6768e623d7e9486b524ddc..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:086038b66aa5fdc1a48f5c9b366bd358080a398470e102309788ea90fd802007 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_002000_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index b6917266ee8490b292e3147d505da50b4b67697d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc30203e67e01d0ec2004e07c4f4aa40bfeb1e856ede30c4f126f2226f626183 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_002362_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 3a848cae8988f684d21b189ba3722152d1e08209..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:478bde5b6ef68865a5b698f3d24ff716442b8a350e42154a5318c57821fbdd55 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/config.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/config.json deleted file mode 100644 index 805b44d5efcb8dd702550036c5cb2d284a1f8c3e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/config.json +++ /dev/null @@ -1,64 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "attention-ablation-v1", - "full-attention", - "fp8" - ] - }, - "config_fingerprint": "9d3bb595965c289e", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1" -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/evals/core.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/evals/core.json deleted file mode 100644 index f336098fd1c151a3cb49d45cdb2ff15ad51923a6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": 0.07475136478377907, - "core_results": { - "hellaswag_zeroshot": 0.2768372893333435, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.08469071239233017, - "arc_easy": 0.3308080732822418, - "arc_challenge": 0.22440272569656372, - "copa": 0.5099999904632568, - "commonsense_qa": 0.315315306186676, - "piqa": 0.5511425137519836, - "openbook_qa": 0.2160000056028366, - "lambada_openai": 0.21793130040168762, - "hellaswag": 0.2775343358516693, - "winograd": 0.6043956279754639, - "winogrande": 0.5011838674545288, - "bigbench_dyck_languages": 0.1080000028014183, - "agi_eval_lsat_ar": 0.25652173161506653, - "bigbench_cs_algorithms": 0.40303027629852295, - "bigbench_operators": 0.05714286118745804, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.01106906309723854, - "coqa": 0.06363522261381149, - "boolq": 0.5709480047225952, - "bigbench_language_identification": 0.25380000472068787 - }, - "centered_results": { - "hellaswag_zeroshot": 0.03578305244445801, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.08469071239233017, - "arc_easy": 0.10774409770965576, - "arc_challenge": -0.034129699071248375, - "copa": 0.019999980926513672, - "commonsense_qa": 0.144144132733345, - "piqa": 0.10228502750396729, - "openbook_qa": -0.04533332586288452, - "lambada_openai": 0.21793130040168762, - "hellaswag": 0.036712447802225746, - "winograd": 0.20879125595092773, - "winogrande": 0.002367734909057617, - "bigbench_dyck_languages": 0.1080000028014183, - "agi_eval_lsat_ar": 0.07065216451883315, - "bigbench_cs_algorithms": 0.40303027629852295, - "bigbench_operators": 0.05714286118745804, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.01106906309723854, - "coqa": 0.06363522261381149, - "boolq": -0.12908419809843363, - "bigbench_language_identification": 0.17909791498425506 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/evals/samples.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/evals/samples.json deleted file mode 100644 index f0c2ae157ca859bc54b233e46e77aa27b47d9cef..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the kingdom of France. The capital of the kingdom of France is" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as that of the sun, and the same as that of the sun" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. I have been in the habit of going to the house of Mr." - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as the opposite of cold. The former is the same as the latter" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the planets, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a dark, dark, and dark color, and I have no doubt that the" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the parts of the body which are in the body, and the" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nTHIS essay was contributed to the \"Introduction to English Literature,\" by the celebrated Mr. A. H.\n\nSmithfield Windle, the author of the Life of Windle, LL.D.\n\nTerritorial Entry Book in Faneuil Hall, Norwich.\n\nBeing a compilation from the Journals of the company which Dunning called the \"Baldive,\" or Committee of the House of Commons, held at a time when the woollen manufacture of New England was taking its full advantage to wear more favourable conditions, it is now very little altered from what it really is.\n\nIn order to have the best assortment of text-books for", - "<|bos|> members were privileged, not asinders, but as guests. Various expressions, for which privileges and responsions would be legal, were freely, but imperfectly disposed of at the time of their introduction, and, so far as it was possible, nothing could serve them except the courtesy of those friendly, courteous, and communicative powers. These statements, we are glad to infer, were understood and appro- 172. priated by friendship. Some leaders of great opinion, such as Joseph Brawns, Van\n\nJoseph reviewers, and all the other leaders of the present war, confined their influence to the side of excluding those who", - "<|bos|>The vision of the man who is mighty to convert brute animals-a man also worthy of the unity which it reveals, wherein he leads a world that is new in its way-Military civilization has attained its unity. Story\n\nApril 10. The vision first revealed to the cison, William\n\nFemale Arnoy as a young warrior who had to find a place in another and more attractive world-a man who formed the purpose of the world of the future-he became first and chiefly interested in the progress of moral and spiritual emancipation which has since made it the aim and object of that great impulse in advance. Essay on the Form of", - "<|bos|>ESLIE'S CHIEFRY.\n\nXX\n\nION.\n\n12 CHANGE DEMOX.\n\nEVE\n\nFAGE OF OPENING THE NILE.\n\nN RATTENOORKS\n\nJEREMIAH PARKER.\n\nNABIKIE BONDLEY.\n\nRAPP SOUTH RADER.\n\nTHE ROCK DAY.\n\nDYCE HIS HOUSE ON A WELL.\n\nPRICES.\n\nPALSTATE OF A CONVERT HOME.\n\nREOTHING\n\nTRANSFORT.\n\nSUPERSET SOUTH AMERICA.\n\n[LIST.]\n\nStraight SPEECH OF CHARLES A. Thacher.\n\nE DELHAND.-THE ROOTS - MEDALS", - "<|bos|>CHAPTER II.\n\nTHE TIDE OF MOUNTAIN PUNISHMENT.\n\nTHE gold has risen by slow stages, but averaged by a slow process. Then it was that the sea, with its red currents running through it, rushed in and pursued its great voyage; and the whole country in full wonder at our eagerness soon seemed to vanish, of the torrents of blood rushing in, on either side, like liquid salt streams [430] ready to overflow. And at last this mixture of wind and tide came into use [441] much as in the ancient days of that ancient city the sea called Adria\n\nCONCLUDED", - "<|bos|>PREFACE\n\nand PERS. likewise. But pictures and pictures of the passions pass into the hands of the beholder after they are painted: at first, but not before, it is the furthest stage of thoughts-catching always that pursuit the quiet enjoyment, and the most immediate order, of nature.\n\nA people so polished that they cannot look at people without perceiving much vanity and often whimsicality, an expression of fine sense, a thousand ways excelling the common language of the middle ages, and that modes of thought are the most subtle, rare, and beautiful that could have been devised.\n\nAnd, lastly, the feelings when painted are", - "<|bos|>Clatim. Sultan\n\nLocker v. Manchester, &c., Co., 34 Fed. 222.\n\n125a. \u00fc 180 a; 3.5 P.\n\n180 a; 7.8 P. b. 181 a; 8.79 P. b. 182; 9.2 P. b. 185a; 10.4 P. b. 187q.\n\n184 a; 11.6 P. b. 187 b; 12.5 P. b. 185 b.\n\n18", - "<|bos|>Lestr Menderella and Victoria.\" \"You know that that parent of ours the conductor was not forgotten in this place.\" \"For many days large numbers of People were detained at Passes.\" \"About four people intervened, chiefly middle-sized disreputable boys, unaccus-\n\ntimed with that kind of respectability which Blot had never reached but in those days when life was so unsatisfactory.\" \"Do you remember having acted as conductor?\" \"I remember easily enough.\" \"Well, I know about very many people.\" \"6 \"Ah, is it true that Representatives are allowed to become securities for themselves?\"\n\nMass. \"Disability" - ] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/evals/val_bpb.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/evals/val_bpb.json deleted file mode 100644 index 9a599e62a64f4de5ba63996a92ad41450547a366..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.0353479590828456 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/run.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/run.json deleted file mode 100644 index 3b2d8ae0bfa1355de190d5b667536837e8d89fc4..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "9d3bb595965c289e", - "wandb_run_id": "73b2700b", - "created_at": 1784316816 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/checkpoints/meta_000021.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/checkpoints/meta_000021.json deleted file mode 100644 index 8415c6a50964d1b2b6e8abe793895dcf379170d2..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/checkpoints/meta_000021.json +++ /dev/null @@ -1,114 +0,0 @@ -{ - "step": 21, - "training_complete": true, - "val_bpb": 0.9112977907020111, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5", - "wandb_run_id": "332d9c75", - "wandb_group": "think-d12", - "wandb_tags": "sft,pre1930-routes,stem5", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/base_checkpoints", - "base_step": 2362, - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/checkpoints", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer", - "resume_from_step": null, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/config.json", - "parent_cumulative_flops": 1.3790348826629898e+18, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "fd70d9abb7eaeffa712c8b9665d7d9e7495e9ee8", - "load_optimizer": 1, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.0, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": -1, - "eval_tokens": 20971520, - "chatcore_every": -1, - "chatcore_max_cat": -1, - "chatcore_max_sample": 24, - "save_every": -1, - "recipe": "pre1930-routes", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-synth-stem5", - "data": { - "recipe": "pre1930-routes", - "stem_reasoning_epochs": 5 - }, - "training": { - "num_iterations": -1, - "device_batch_size": 8, - "eval_every": -1, - "chatcore_every": -1, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "pre1930-routes", - "stem5" - ] - }, - "config_fingerprint": "0e7e06cac4e1dfb1", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "0e7e06cac4e1dfb1" - }, - "loop_state": { - "step": 21, - "total_training_time": 33.068419218063354, - "min_val_bpb": 0.9112977907020111, - "smooth_train_loss": 1.645360864787687, - "mfu": 62.1129491496942, - "tok_per_sec": 174024, - "stage_training_flops": 1.2260682699374592e+16, - "inherited_parent_flops": 1.3790348826629898e+18, - "cumulative_pipeline_training_flops": 1.3912955653623644e+18 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/checkpoints/model_000021.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/checkpoints/model_000021.pt deleted file mode 100644 index 4cf766ea5946859a6cdc68f642b422435875c84a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/checkpoints/model_000021.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3fae981d730dbf371edb3d15aa2a020f9eb537832c56ce2994ad4c2146e85aac -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/checkpoints/optim_000021_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/checkpoints/optim_000021_rank0.pt deleted file mode 100644 index a9f9fe7a6a459159c4e2114a4447ba0ac26ee390..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/checkpoints/optim_000021_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:27e05c2f420da92d2c012237a58c1cb244a8392321e5bf99e2254cb16937b013 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/config.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/config.json deleted file mode 100644 index 87940571e48485bc5be39f1fede643a2bb88d6b0..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/config.json +++ /dev/null @@ -1,32 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-synth-stem5", - "data": { - "recipe": "pre1930-routes", - "stem_reasoning_epochs": 5 - }, - "training": { - "num_iterations": -1, - "device_batch_size": 8, - "eval_every": -1, - "chatcore_every": -1, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "pre1930-routes", - "stem5" - ] - }, - "config_fingerprint": "0e7e06cac4e1dfb1", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5" -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/run.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/run.json deleted file mode 100644 index 08da5f6e1d17e1c9f959a644e19df4fff299f8ce..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5", - "stage": "sft", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "parent_experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "parent_checkpoint_step": null, - "branch_parent_step": null, - "config_fingerprint": "0e7e06cac4e1dfb1", - "wandb_run_id": "332d9c75", - "created_at": 1785518357 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/summary.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/summary.json deleted file mode 100644 index 9fd85fa3cd4b765bcb1775ef550c4c8cdfcf225b..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5/summary.json +++ /dev/null @@ -1,19 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5", - "stage": "sft", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "parent_experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "parent_checkpoint_step": null, - "step": 21, - "stage_training_flops": 1.2260682699374592e+16, - "inherited_parent_flops": 1.3790348826629898e+18, - "cumulative_pipeline_training_flops": 1.3912955653623644e+18, - "chatcore_metric": null, - "chat_results": null, - "training_time_seconds": 33.068419218063354, - "config_fingerprint": "0e7e06cac4e1dfb1", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "fd70d9abb7eaeffa712c8b9665d7d9e7495e9ee8", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/332d9c75", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/sft/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1-pre1930-synth-stem5" -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/summary.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/summary.json deleted file mode 100644 index 4bc7c8744379cebbd9c9c771b88ef6f97596318b..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/summary.json +++ /dev/null @@ -1,91 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean-1930s", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.089810417727749, - "minimum_sampled_val_bpb": 1.089810417727749, - "full_val_bpb": 1.0353479590828456, - "core_metric": 0.07475136478377907, - "centered_results": { - "hellaswag_zeroshot": 0.03578305244445801, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.08469071239233017, - "arc_easy": 0.10774409770965576, - "arc_challenge": -0.034129699071248375, - "copa": 0.019999980926513672, - "commonsense_qa": 0.144144132733345, - "piqa": 0.10228502750396729, - "openbook_qa": -0.04533332586288452, - "lambada_openai": 0.21793130040168762, - "hellaswag": 0.036712447802225746, - "winograd": 0.20879125595092773, - "winogrande": 0.002367734909057617, - "bigbench_dyck_languages": 0.1080000028014183, - "agi_eval_lsat_ar": 0.07065216451883315, - "bigbench_cs_algorithms": 0.40303027629852295, - "bigbench_operators": 0.05714286118745804, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.01106906309723854, - "coqa": 0.06363522261381149, - "boolq": -0.12908419809843363, - "bigbench_language_identification": 0.17909791498425506 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the kingdom of France. The capital of the kingdom of France is" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as that of the sun, and the same as that of the sun" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. I have been in the habit of going to the house of Mr." - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as the opposite of cold. The former is the same as the latter" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the planets, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a dark, dark, and dark color, and I have no doubt that the" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the parts of the body which are in the body, and the" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nTHIS essay was contributed to the \"Introduction to English Literature,\" by the celebrated Mr. A. H.\n\nSmithfield Windle, the author of the Life of Windle, LL.D.\n\nTerritorial Entry Book in Faneuil Hall, Norwich.\n\nBeing a compilation from the Journals of the company which Dunning called the \"Baldive,\" or Committee of the House of Commons, held at a time when the woollen manufacture of New England was taking its full advantage to wear more favourable conditions, it is now very little altered from what it really is.\n\nIn order to have the best assortment of text-books for", - "<|bos|> members were privileged, not asinders, but as guests. Various expressions, for which privileges and responsions would be legal, were freely, but imperfectly disposed of at the time of their introduction, and, so far as it was possible, nothing could serve them except the courtesy of those friendly, courteous, and communicative powers. These statements, we are glad to infer, were understood and appro- 172. priated by friendship. Some leaders of great opinion, such as Joseph Brawns, Van\n\nJoseph reviewers, and all the other leaders of the present war, confined their influence to the side of excluding those who", - "<|bos|>The vision of the man who is mighty to convert brute animals-a man also worthy of the unity which it reveals, wherein he leads a world that is new in its way-Military civilization has attained its unity. Story\n\nApril 10. The vision first revealed to the cison, William\n\nFemale Arnoy as a young warrior who had to find a place in another and more attractive world-a man who formed the purpose of the world of the future-he became first and chiefly interested in the progress of moral and spiritual emancipation which has since made it the aim and object of that great impulse in advance. Essay on the Form of", - "<|bos|>ESLIE'S CHIEFRY.\n\nXX\n\nION.\n\n12 CHANGE DEMOX.\n\nEVE\n\nFAGE OF OPENING THE NILE.\n\nN RATTENOORKS\n\nJEREMIAH PARKER.\n\nNABIKIE BONDLEY.\n\nRAPP SOUTH RADER.\n\nTHE ROCK DAY.\n\nDYCE HIS HOUSE ON A WELL.\n\nPRICES.\n\nPALSTATE OF A CONVERT HOME.\n\nREOTHING\n\nTRANSFORT.\n\nSUPERSET SOUTH AMERICA.\n\n[LIST.]\n\nStraight SPEECH OF CHARLES A. Thacher.\n\nE DELHAND.-THE ROOTS - MEDALS", - "<|bos|>CHAPTER II.\n\nTHE TIDE OF MOUNTAIN PUNISHMENT.\n\nTHE gold has risen by slow stages, but averaged by a slow process. Then it was that the sea, with its red currents running through it, rushed in and pursued its great voyage; and the whole country in full wonder at our eagerness soon seemed to vanish, of the torrents of blood rushing in, on either side, like liquid salt streams [430] ready to overflow. And at last this mixture of wind and tide came into use [441] much as in the ancient days of that ancient city the sea called Adria\n\nCONCLUDED", - "<|bos|>PREFACE\n\nand PERS. likewise. But pictures and pictures of the passions pass into the hands of the beholder after they are painted: at first, but not before, it is the furthest stage of thoughts-catching always that pursuit the quiet enjoyment, and the most immediate order, of nature.\n\nA people so polished that they cannot look at people without perceiving much vanity and often whimsicality, an expression of fine sense, a thousand ways excelling the common language of the middle ages, and that modes of thought are the most subtle, rare, and beautiful that could have been devised.\n\nAnd, lastly, the feelings when painted are", - "<|bos|>Clatim. Sultan\n\nLocker v. Manchester, &c., Co., 34 Fed. 222.\n\n125a. \u00fc 180 a; 3.5 P.\n\n180 a; 7.8 P. b. 181 a; 8.79 P. b. 182; 9.2 P. b. 185a; 10.4 P. b. 187q.\n\n184 a; 11.6 P. b. 187 b; 12.5 P. b. 185 b.\n\n18", - "<|bos|>Lestr Menderella and Victoria.\" \"You know that that parent of ours the conductor was not forgotten in this place.\" \"For many days large numbers of People were detained at Passes.\" \"About four people intervened, chiefly middle-sized disreputable boys, unaccus-\n\ntimed with that kind of respectability which Blot had never reached but in those days when life was so unsatisfactory.\" \"Do you remember having acted as conductor?\" \"I remember easily enough.\" \"Well, I know about very many people.\" \"6 \"Ah, is it true that Representatives are allowed to become securities for themselves?\"\n\nMass. \"Disability" - ], - "training_time_seconds": 2747.30322766304, - "stage_training_flops": 1.3790348826629898e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.3790348826629898e+18, - "config_fingerprint": "9d3bb595965c289e", - "git_commit_sha": "fd70d9abb7eaeffa712c8b9665d7d9e7495e9ee8", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/73b2700b", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1", - "dataset_fingerprint": "0db4605cfe3a7eac", - "tokenizer_fingerprint": "21e99d5cdeeaa660" -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer/token_bytes.pt b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer/tokenizer.pkl b/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-full-fulltok-ablation-v1/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_000500.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_000500.json deleted file mode 100644 index 94b33333495b63f89661f3b35a1d9736afad22f5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,147 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "val_bpb": 1.3388062223680175, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "wandb_run_id": "1528c501", - "wandb_group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,attention-ablation-v1,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "attention-ablation-v1", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "b2b7a42993186da5", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "b2b7a42993186da5" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3388062223680175, - "smooth_train_loss": 3.697135257265452, - "total_training_time": 529.8762292861938, - "stage_training_flops": 225125685264384000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 225125685264384000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_001000.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_001000.json deleted file mode 100644 index 3928cf6a33941108da94342e3dc4aaac807f1b5b..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,147 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "val_bpb": 1.232943966369166, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "wandb_run_id": "1528c501", - "wandb_group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,attention-ablation-v1,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "attention-ablation-v1", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "b2b7a42993186da5", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "b2b7a42993186da5" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.232943966369166, - "smooth_train_loss": 3.515408592820662, - "total_training_time": 1071.7539193630219, - "stage_training_flops": 450251370528768000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 450251370528768000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_001500.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_001500.json deleted file mode 100644 index e3f31ff300dc12b88e88a45aefad63d085b48761..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,147 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "val_bpb": 1.1509991902459196, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "wandb_run_id": "1528c501", - "wandb_group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,attention-ablation-v1,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "attention-ablation-v1", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "b2b7a42993186da5", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "b2b7a42993186da5" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.1509991902459196, - "smooth_train_loss": 3.402263473061546, - "total_training_time": 1613.613620042801, - "stage_training_flops": 675377055793152000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 675377055793152000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_002000.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_002000.json deleted file mode 100644 index f2ed72554a14e215c6f4b0e221cc2e39ffbf8dea..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,147 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "val_bpb": 1.1138402204171254, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "wandb_run_id": "1528c501", - "wandb_group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,attention-ablation-v1,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "attention-ablation-v1", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "b2b7a42993186da5", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "b2b7a42993186da5" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.1138402204171254, - "smooth_train_loss": 3.335568630966098, - "total_training_time": 2155.4914762973785, - "stage_training_flops": 900502741057536000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 900502741057536000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_002362.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_002362.json deleted file mode 100644 index 661ac6ff883ea5052e245450711a365af41e5720..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,147 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "val_bpb": 1.09204857698696, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "wandb_run_id": "1528c501", - "wandb_group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,attention-ablation-v1,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "attention-ablation-v1", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "b2b7a42993186da5", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "b2b7a42993186da5" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": 1.09204857698696, - "smooth_train_loss": 3.137487900090049, - "total_training_time": 2547.9463839530945, - "stage_training_flops": 1063493737188950016, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1063493737188950016 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_000500.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_000500.pt deleted file mode 100644 index deb1c4c403877355afe30b26c3e5c8cc49ce4f90..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7fa951ce068a1afecc753d53d603601b76d79a3b0450e27ad87d80b79d030c45 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_001000.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_001000.pt deleted file mode 100644 index 75eac5cd3ac4d45315a06491a6d7329994d968db..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e89c40c8e8c831f2b0e9e4e152b98519c9e39f0096b65d34c3dcb18da7419b5f -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_001500.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_001500.pt deleted file mode 100644 index 14cc7c365def29144921f701961c7f64d7f4f20e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:04f941a4497c06fe8a7915babe96271d21742a4c7c7d81c84ef456831fec6820 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_002000.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_002000.pt deleted file mode 100644 index 7603ff8f49ed5be0a3a5f1b9f168ff30387d3aef..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8db77b39d2492a0625d5252493816b36affe1a3b71cfe913d13e57dab33a88f5 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_002362.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_002362.pt deleted file mode 100644 index 4e23b95850ae6d44b7abfa6e488a0e8737f701ae..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:07a30cd3d55ad8c34531d38822c0e49eb80ce81acdca53ee39ef4095b098b89c -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_000500_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 43968ab486eb98708ad8cc4709469213040ff872..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1241b441b7b3c4b7c6670e0448f1c643ef76b2e29aecb20bbd8e9b47b4c741c0 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_001000_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 2e5fc814f982e311d0ab1b65741b05314c6dd2f6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d53d042be62262610cd82685ab9cb7988349f902de2a662cbcc7bd3b6be358ad -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_001500_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index b9e99a3abfa6064953f06e3c511b40f127fc4f04..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f7e5e0fb40d9e8da1088b7eb13779d3b40c35d78d2f92c598d77a5e870401fa2 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_002000_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index d8a951c7ca9e7c6be3b9ce4127bc15bc818ae32e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8789a8db6987ebaf83061895c81115cd55de86023eaa2dc9f3cf43068da1abe6 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_002362_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 1f91834de9820ac617717768274f8ce44f4761f1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:73905482cf6275645c38ef7abb576f97b07a3b694b3d50f8f083ccbe955321bd -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/config.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/config.json deleted file mode 100644 index 206aa1fd4c2cfb3437d45fee41c15a6cfe2c9a8c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/config.json +++ /dev/null @@ -1,64 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "group": "clean1930s-d12-ctx4096-attention-ablation-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "attention-ablation-v1", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "b2b7a42993186da5", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1" -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/evals/samples.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/evals/samples.json deleted file mode 100644 index e8abb0c133117628d739cc248892daff0f95e4b2..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the kingdom of France, and the capital of the kingdom of England" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the gold of the earth. The symbol of the gold of the" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If yesterday is Saturday, then tomorrow will be Saturday. If yesterday is" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as the opposite of cold. The latter is the same as the former" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the planets, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the red, and I have a great many other colors. I have a great" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of days in the year, and x is the number of days in" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nTHIS essay was contributed to the \"Preliminary.\" Its author says \"This work is undertaken for candidates who love the genius and industry of the manufacturing people of the United States. Among those who love the Cliff institutions in Faneuil Hall, Fort Hall, Manchester, and Price's ostentatious company, Dunning, Upton, Tecumseh, Whittaker & Co. Stockport, Stockport, Rutland, Buckingham, Cole, & Co., & Co., are included.\"* The Address to the People of Faneuil Hall, Garrisonville, is published, is by Mr. Hoikum, and", - "<|bos|>PREFACE.\n\nTHE record of the Scotch courts of justice from 1807 to 1812 is published with the assistance of notes by\n\nGeorge W. Ellis, of New South Wales. The studies prosecuted in Scottish courts during the past year were 62 Scotch organized courts in the county of Cumberland, the Causes of the\n\nCauses of the first district court court court courts court courts court courts courts court court courts court courts courts court courts court courts court court courts court court courts courts court courts court courts court courts court courts court review court courts court court court courts court courts court court court courts courts court courts court courts court courts court court court", - "<|bos|>The vision ended with an enumeration of Wilson's impressions on Dyke's memory.\n\nA little sequel then went forth, wherein I find a world of tender notes in Life's Present Space, delivered upon the Religious Assemblics, now greatly refreshed and strengthened. These are copied c. V.\n\nINTRODUCTION OF THE ATHELLO'S FRIENDShip.\n\n105 upon the virtues' and what they are, which but formed the background of the diverse portraits conceived for each age. Long ago that painter died, like Saul and\n\nJonathan, among theocrats. On his death cancel that great travail in England. Certes, the", - "<|bos|> ministered to 350,752 zeal for the one great institution of learning, or Christian faith, which Christ died to be remembered by his disciples. William Rutherford, in 1840, was a farmer; and the \"Missionary Bldg. and Farmer Steenin, in their investigations as to the rate of salary allowed in England 50 years ago, have shown, that charges of this kind (which consist chiefly of percentages) are too high. Even 3 years last past beer-and-day rate would consequently be but little in vogue this season.\n\nWere we to have another Sunday-school at -2-y", - "<|bos|>eed\n\nNew York paid them her separate yearly pension of \u00a33,700 a year. It would be quite sufficient to hustle the\n\nCrown publicly against the attacks of such visitants. But to imprison her was still not allowed; she would break down. The subsidy in her stead was at our disposal. In April, 1916, certain British subjects started to attack America, with the view of bringing about a peace with Madagascar. And this, on their part, was successfully accomplished.\n\nDivers wooden beams were shot across the country. Since 1913 the preparation of A. H. Wright's paper for", - "<|bos|>PREFACE\n\n771. The property when pictures and pictures was being discussed at the hotel, and party papers, letters, etiquette, the French management, and of course the general reception of the day; and thoughts of it always took place in household affairs, and on most public questions, and by fair and regular means it was not unfrequently found to match Fish's bill book and \"possibly \" an opportunity of getting expressions of reform from her chairman, and among many other ceremonials in her favour, and Christian Orientalizing the German people, and breaking the two latter semi-faces.\n\nIt sometimes happened, as the feelings of the audience", - "<|bos|>) have they not repeated in the Bible the same injunction, \"join unto all the rest of the world.\" What do these seven epistles contain? It 1\n\nAnd do metaphorical language not admit of a similar application? smell no of blood\" it is simply and solely by thinking they are convenient.\n\nBut are you thinking about blood? Are your expenses great? Is there nothing but blood which you use?\n\nDo you read the book of which I write ?\"\n\nIs there nothing but rosewood? Is your house full for gardening? Is there fine linen for bloomery? If so, how much more for open doors? But how", - "<|bos|>PREFACE.\n\nbut Mendez considers himself rewarded in a large domain by that freedom of epistolary intercourse from which he had so long endured a painful bondage.\n\nHe felt that neither the virtue of his literary embrasure nor the man would bear disreputation so long as he lived.\n\nThat he should have seized upon that which constituted the most consoling but the direst part of his destiny may be doubted; though he well knew that Mendez looked forward with an agonizing pang to the end. But he did not confine himself to the object of discipline; he aspired to a familiar intercourse with his native land. And yet perhaps even" - ] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/evals/val_bpb.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/evals/val_bpb.json deleted file mode 100644 index b8f3d9844a0075cff3bb0d0487b33db49732e8c5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.0373025644430633 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/run.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/run.json deleted file mode 100644 index 6917f8ebff63c2d8d95dbdbe097b22767b686453..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "b2b7a42993186da5", - "wandb_run_id": "1528c501", - "created_at": 1784313993 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/summary.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/summary.json deleted file mode 100644 index fa89d86def6317fefba6d18b3d9c8bdb40c6de5b..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/summary.json +++ /dev/null @@ -1,69 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean-1930s", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.09204857698696, - "minimum_sampled_val_bpb": 1.09204857698696, - "full_val_bpb": 1.0373025644430633, - "core_metric": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the kingdom of France, and the capital of the kingdom of England" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the gold of the earth. The symbol of the gold of the" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If yesterday is Saturday, then tomorrow will be Saturday. If yesterday is" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as the opposite of cold. The latter is the same as the former" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the planets, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the red, and I have a great many other colors. I have a great" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of days in the year, and x is the number of days in" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nTHIS essay was contributed to the \"Preliminary.\" Its author says \"This work is undertaken for candidates who love the genius and industry of the manufacturing people of the United States. Among those who love the Cliff institutions in Faneuil Hall, Fort Hall, Manchester, and Price's ostentatious company, Dunning, Upton, Tecumseh, Whittaker & Co. Stockport, Stockport, Rutland, Buckingham, Cole, & Co., & Co., are included.\"* The Address to the People of Faneuil Hall, Garrisonville, is published, is by Mr. Hoikum, and", - "<|bos|>PREFACE.\n\nTHE record of the Scotch courts of justice from 1807 to 1812 is published with the assistance of notes by\n\nGeorge W. Ellis, of New South Wales. The studies prosecuted in Scottish courts during the past year were 62 Scotch organized courts in the county of Cumberland, the Causes of the\n\nCauses of the first district court court court courts court courts court courts courts court court courts court courts courts court courts court courts court court courts court court courts courts court courts court courts court courts court courts court review court courts court court court courts court courts court court court courts courts court courts court courts court courts court court court", - "<|bos|>The vision ended with an enumeration of Wilson's impressions on Dyke's memory.\n\nA little sequel then went forth, wherein I find a world of tender notes in Life's Present Space, delivered upon the Religious Assemblics, now greatly refreshed and strengthened. These are copied c. V.\n\nINTRODUCTION OF THE ATHELLO'S FRIENDShip.\n\n105 upon the virtues' and what they are, which but formed the background of the diverse portraits conceived for each age. Long ago that painter died, like Saul and\n\nJonathan, among theocrats. On his death cancel that great travail in England. Certes, the", - "<|bos|> ministered to 350,752 zeal for the one great institution of learning, or Christian faith, which Christ died to be remembered by his disciples. William Rutherford, in 1840, was a farmer; and the \"Missionary Bldg. and Farmer Steenin, in their investigations as to the rate of salary allowed in England 50 years ago, have shown, that charges of this kind (which consist chiefly of percentages) are too high. Even 3 years last past beer-and-day rate would consequently be but little in vogue this season.\n\nWere we to have another Sunday-school at -2-y", - "<|bos|>eed\n\nNew York paid them her separate yearly pension of \u00a33,700 a year. It would be quite sufficient to hustle the\n\nCrown publicly against the attacks of such visitants. But to imprison her was still not allowed; she would break down. The subsidy in her stead was at our disposal. In April, 1916, certain British subjects started to attack America, with the view of bringing about a peace with Madagascar. And this, on their part, was successfully accomplished.\n\nDivers wooden beams were shot across the country. Since 1913 the preparation of A. H. Wright's paper for", - "<|bos|>PREFACE\n\n771. The property when pictures and pictures was being discussed at the hotel, and party papers, letters, etiquette, the French management, and of course the general reception of the day; and thoughts of it always took place in household affairs, and on most public questions, and by fair and regular means it was not unfrequently found to match Fish's bill book and \"possibly \" an opportunity of getting expressions of reform from her chairman, and among many other ceremonials in her favour, and Christian Orientalizing the German people, and breaking the two latter semi-faces.\n\nIt sometimes happened, as the feelings of the audience", - "<|bos|>) have they not repeated in the Bible the same injunction, \"join unto all the rest of the world.\" What do these seven epistles contain? It 1\n\nAnd do metaphorical language not admit of a similar application? smell no of blood\" it is simply and solely by thinking they are convenient.\n\nBut are you thinking about blood? Are your expenses great? Is there nothing but blood which you use?\n\nDo you read the book of which I write ?\"\n\nIs there nothing but rosewood? Is your house full for gardening? Is there fine linen for bloomery? If so, how much more for open doors? But how", - "<|bos|>PREFACE.\n\nbut Mendez considers himself rewarded in a large domain by that freedom of epistolary intercourse from which he had so long endured a painful bondage.\n\nHe felt that neither the virtue of his literary embrasure nor the man would bear disreputation so long as he lived.\n\nThat he should have seized upon that which constituted the most consoling but the direst part of his destiny may be doubted; though he well knew that Mendez looked forward with an agonizing pang to the end. But he did not confine himself to the object of discipline; he aspired to a familiar intercourse with his native land. And yet perhaps even" - ], - "training_time_seconds": 2547.9463839530945, - "stage_training_flops": 1.06349373718895e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.06349373718895e+18, - "config_fingerprint": "b2b7a42993186da5", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/1528c501", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1", - "dataset_fingerprint": "0db4605cfe3a7eac", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "unique_train_tokens": 0 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer/token_bytes.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer/tokenizer.pkl b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-ablation-v1/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_000500.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_000500.json deleted file mode 100644 index 88494617d3b7d1c0073098f4957218d77197ae2d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,161 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.340710622409166, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "wandb_run_id": "dd182242", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,current-code-baseline,seed42", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": null, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "current-code-baseline", - "seed42" - ] - }, - "config_fingerprint": "b389d34fe603d8f3", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "b389d34fe603d8f3" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.340710622409166, - "smooth_train_loss": 3.699678794303341, - "total_training_time": 527.217556476593, - "stage_start_step": 0, - "stage_training_flops": 225125685264384000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2.25125685264384e+17 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_001000.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_001000.json deleted file mode 100644 index c73804a90519745bce827188e19f968415b2875a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,161 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.2318290352476646, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "wandb_run_id": "dd182242", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,current-code-baseline,seed42", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": null, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "current-code-baseline", - "seed42" - ] - }, - "config_fingerprint": "b389d34fe603d8f3", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "b389d34fe603d8f3" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2318290352476646, - "smooth_train_loss": 3.516440409823289, - "total_training_time": 1065.5147593021393, - "stage_start_step": 0, - "stage_training_flops": 450251370528768000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 4.50251370528768e+17 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_001500.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_001500.json deleted file mode 100644 index 964eb90ce91be860ae7a0f3bdc7df5ee03276151..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,161 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.149810760084606, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "wandb_run_id": "dd182242", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,current-code-baseline,seed42", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": null, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "current-code-baseline", - "seed42" - ] - }, - "config_fingerprint": "b389d34fe603d8f3", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "b389d34fe603d8f3" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.149810760084606, - "smooth_train_loss": 3.402021092193979, - "total_training_time": 1603.780502319336, - "stage_start_step": 0, - "stage_training_flops": 675377055793152000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 6.75377055793152e+17 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_002000.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_002000.json deleted file mode 100644 index a62f0550339c9d055185db54450c207bf3ade26e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,161 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.1138188516867007, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "wandb_run_id": "dd182242", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,current-code-baseline,seed42", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": null, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "current-code-baseline", - "seed42" - ] - }, - "config_fingerprint": "b389d34fe603d8f3", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "b389d34fe603d8f3" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.1138188516867007, - "smooth_train_loss": 3.3352839560698695, - "total_training_time": 2142.2500989437103, - "stage_start_step": 0, - "stage_training_flops": 900502741057536000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 9.00502741057536e+17 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_002362.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_002362.json deleted file mode 100644 index 2436b6899e9a97733035345eab9f884e80c402bf..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,161 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.0917705486702252, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "wandb_run_id": "dd182242", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,current-code-baseline,seed42", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": null, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "current-code-baseline", - "seed42" - ] - }, - "config_fingerprint": "b389d34fe603d8f3", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "b389d34fe603d8f3" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": 1.0917705486702252, - "smooth_train_loss": 3.138216938328097, - "total_training_time": 2532.0307886600494, - "stage_start_step": 0, - "stage_training_flops": 1063493737188950016, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.06349373718895e+18 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_000500.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_000500.pt deleted file mode 100644 index d8613958f0070f20b4302b01ce2a73bd504d24db..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2b5618c95938c3ffaed59cafc1817f4785748f0993dc56080c8a1adf980bbf3f -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_001000.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_001000.pt deleted file mode 100644 index a13e6ad641fe851d006e330d14faacf8f7710c97..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bfd8c9fa9b4cdcc67b3b1f21100bcac0e24b245573070b81a2e2f46e5a1ca288 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_001500.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_001500.pt deleted file mode 100644 index e313b2c3420c1eb1b8dd69a85420600d354ea10e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7efa0c331acb4544d2fbf7d3cb43df147912bd75d5ab795b537395881780bf07 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_002000.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_002000.pt deleted file mode 100644 index fb4bda0c0db5971b0b660974d24255daca75e776..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:81d36859bf90675e49623612243c416b178d76a948ee3c430d4924f33c26897b -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_002362.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_002362.pt deleted file mode 100644 index 380f0e1548f98a6a87d93b6cb61722bd9f122290..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a51d4b72445597a7798b15fa5d7133a16a69f7787cda73fb57f2776adda55cef -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_000500_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index efd7d94dc795b207e0f6dcae53077ed27c712175..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1328c5fac82a4f04f8d848a8fa242cfaa574aab99cdfe7f41b70d89ff73abdba -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_001000_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index d4998e6824733bd3f7a36dc2ccceefda5a875083..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5bfe23e6eb60170c5f9f75e19f8247102a5cb51a274cb31b128e0295f4492482 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_001500_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 3fa076e0d61e66ab49ad7393ded2166dc834b4d3..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5c6bab8cc63969fa127be0f034efc804c532b69cc2fad979359d76f4b0876fdc -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_002000_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index bae23b0dc47093c13c8a4038aaf95064c47c5b6c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:dd1b41090682534e2879ea0e2dacce9c629d493a07a1615b31004690a9d74f03 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_002362_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index e3416d8745ff936a95f088f254d0383ee338a4c1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e3b6033cc6d317b1a2b337e91b1cc3e4f73250f88c627e29b8da229889932196 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/config.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/config.json deleted file mode 100644 index b3c59c13a7e460dc08692ec39e251aa67142b577..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/config.json +++ /dev/null @@ -1,68 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "current-code-baseline", - "seed42" - ] - }, - "config_fingerprint": "b389d34fe603d8f3", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1" -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/evals/samples.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/evals/samples.json deleted file mode 100644 index d687336b1d8afcecc8fb3487cd32781867dbbc41..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the French nation, and the capital of the French nation, and" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as the symbol of silver, and the symbol of gold is the same" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and the same day will be Saturday, and the same day will be" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as the opposite of cold, and the same as the opposite of hot" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: 1. The sun, 2. The moon, 3. The" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a deep blue, with a deep blue border, and a deep blue border." - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the same as x, and x is the same as x, and x is" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.--S. FLETCHER LONDON, AND A NEW STUDY FOR THE SERMONS AND SERMONS, LECTURES, AND SERMONS, X-31.\n\nReferred to the Clergy of Gloucester Fair Art Training School Clauses\n\nS F. FAITHFULLY anticipating the positions which they are to occupy.-now that Dunning, Dean of Westminster, has removed from their side, namely, to Advent Sunday-School, and did even more for the cause of Education, the Burgesses have more openly expressed a wish to remain obscure.-Walker's Per.\n\nLamens, p. xvii.\n\nAfter he called on me at the University", - "<|bos|>PREFACE.\n\nTHE LIBRARY OF ACOPYRIGHT AND FIRST CONTINENT INST.\n\nM.A.\n\nCONTENTS.\n\nxix\n\nastes forlorn deliver\n\nthe fear or dan- trance to the work of ve- ners.\n\nfor it under the w 62 artist's motive lies; he gave a hand couric Woman the first couple- one leaf, weeded two other- tyred five-teixalys with emphasis. or that by friendship made two-cywork, no hand guards; the precious names Minister hand method technical review of machinery, all the idiot's necessary assistants\n\nThe handbook of defenceominal arguments over means in my", - "<|bos|>AUXE\n\nDISCOURSE XII.\n\n\"Gray Dyfe's \"Primer of Poetry.\n\nHERP, I-XXVI. \"The Marquis de Faug\u00e8res, who made the longest ode of verse\n\nApril 10. Four Years. \"LETTER XIII.\n\n155 William X. to His Sister, as it was sold at St. Petersburg, (?) in 1837. \"DEAR JOHN,-do you, I, mean to advise me through your father and your cousin Robert, how much moral and spiritual worth the person of Joseph Bonaparte may make upon him? that great intellect in England is as dear to me as", - "<|bos|> ministered, 350, 353, 355; marvelled at it till it stirred up heartfelt gratitude, 353; at\n\nMerton in the East, 40, 41, 42, 51; Burbow, the minister of the cross, 41 depicted as one whose religio powers harmonized with 35 Entertainment of this symbolism means, as I believe it would tend to debauch the thoughts and organize tendencies of a people to sin, and furnish an environment consequently personal to itself, but this avers divine authority, as developed in lies;\n\n692;", - "<|bos|>CHAPTER XIII.\n\nTHE TIDE WOMEN AND THE HEART OF THE BIBLE.\n\nAr the time of our Lord's Passion, a tenant of\n\nCrowne herself, a friendly, find, was found to be alive.\n\nAnd in truth, this great lady had madd' unto her taper.\n\nFor our Lord being dead, to his sence the sole effect was, to be dead.\n\nFrom her hand have we no reason to pray.\n\nSo am I, poore, married, or dead.\n\nthank\n\nDivers times I've much doubted whether the ghost of\n\nHobany resumed her hold, either in the Church where his", - "<|bos|>PREFACE XIII, PERS. 92. Besides these and other volumes of individuals, there are a few essays dealing respectively with political, the French Revolution, and the Confederate and Bankrupt countries,1 in general thoughts, words, and ways.\n\nWhen I have to thank most indulgent Heaven, summoned by fair profit the sowing of that mind concerning every subject which is at stake, and found it in that man who needs criticise expressions, reformations, improvisements, and principles insufficient in every one of them, it is the intention of this chapter to explain the principles and all manner of civilities, whether in political and social life,", - "<|bos|>) have published a\n\nHEREDITY OF HOMESTEAD DICTION\n\nThe proper pronunciation of the title, Bundl, and its necessary 1 a ao, band, ning from b for a. for, 17 it b simply and the na = 1 m\n\n79ly arrived at, everyday are always :\n\n23! @2 o2 b\n\nisy,\n\nThe letters and dulces of the Author in this and in the other statements, are therefore more warranted in being either for m, of, 1 or for bb, then jv\u00f4vi\u00e9 for m.\n\nThe sounds o", - "<|bos|>osso caso, e qu-cell-\n\nLive out of act e assumes he at to act of vindi-neling in the ni sa which preceded v, which was not more than vi h age and contemporaneous time, yet\n\n(rule Mn 4\n\nJOHN PIERRE 209\n\nmore to multiply v that life posses so HENT INTIF\n\n1802] (he adds, al pacity, ju \u200b1 n 227 25 194 \u664a\u200b\u9ac3 \u200b01 z\n\nSecondly, command your temper with love let the nag in to hehey n" - ] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/evals/val_bpb.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/evals/val_bpb.json deleted file mode 100644 index d3763de408301bea470a103fcb6391150ed31b45..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/evals/val_bpb.json +++ /dev/null @@ -1,94 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val_per_position": [ - { - "start": 0, - "end": 256, - "bpb": 1.1280390202815782 - }, - { - "start": 256, - "end": 512, - "bpb": 1.0668321602380306 - }, - { - "start": 512, - "end": 768, - "bpb": 1.0538636279911089 - }, - { - "start": 768, - "end": 1024, - "bpb": 1.0433245295237108 - }, - { - "start": 1024, - "end": 1280, - "bpb": 1.037325000637822 - }, - { - "start": 1280, - "end": 1536, - "bpb": 1.035617094387613 - }, - { - "start": 1536, - "end": 1792, - "bpb": 1.031107901633012 - }, - { - "start": 1792, - "end": 2048, - "bpb": 1.0284495007735763 - }, - { - "start": 2048, - "end": 2304, - "bpb": 1.0252337585887044 - }, - { - "start": 2304, - "end": 2560, - "bpb": 1.0209599977356112 - }, - { - "start": 2560, - "end": 2816, - "bpb": 1.0234279467336143 - }, - { - "start": 2816, - "end": 3072, - "bpb": 1.0242444945017173 - }, - { - "start": 3072, - "end": 3328, - "bpb": 1.0204794517385218 - }, - { - "start": 3328, - "end": 3584, - "bpb": 1.018473960788501 - }, - { - "start": 3584, - "end": 3840, - "bpb": 1.0170625289877553 - }, - { - "start": 3840, - "end": 4096, - "bpb": 1.0161504365408973 - } - ], - "val": 1.0369184312160626 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/run.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/run.json deleted file mode 100644 index 0baf310f643e09c246106b078517949b09ce1d8f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "branch_parent_step": null, - "config_fingerprint": "b389d34fe603d8f3", - "wandb_run_id": "dd182242", - "created_at": 1785980999 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/summary.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/summary.json deleted file mode 100644 index 850af5a3b6a28380ae4c64fb73f1ead6d0bc5e1c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/summary.json +++ /dev/null @@ -1,70 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean-1930s", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.0917705486702252, - "minimum_sampled_val_bpb": 1.0917705486702252, - "full_val_bpb": 1.0369184312160626, - "core_metric": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the French nation, and the capital of the French nation, and" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as the symbol of silver, and the symbol of gold is the same" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and the same day will be Saturday, and the same day will be" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as the opposite of cold, and the same as the opposite of hot" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: 1. The sun, 2. The moon, 3. The" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a deep blue, with a deep blue border, and a deep blue border." - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the same as x, and x is the same as x, and x is" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.--S. FLETCHER LONDON, AND A NEW STUDY FOR THE SERMONS AND SERMONS, LECTURES, AND SERMONS, X-31.\n\nReferred to the Clergy of Gloucester Fair Art Training School Clauses\n\nS F. FAITHFULLY anticipating the positions which they are to occupy.-now that Dunning, Dean of Westminster, has removed from their side, namely, to Advent Sunday-School, and did even more for the cause of Education, the Burgesses have more openly expressed a wish to remain obscure.-Walker's Per.\n\nLamens, p. xvii.\n\nAfter he called on me at the University", - "<|bos|>PREFACE.\n\nTHE LIBRARY OF ACOPYRIGHT AND FIRST CONTINENT INST.\n\nM.A.\n\nCONTENTS.\n\nxix\n\nastes forlorn deliver\n\nthe fear or dan- trance to the work of ve- ners.\n\nfor it under the w 62 artist's motive lies; he gave a hand couric Woman the first couple- one leaf, weeded two other- tyred five-teixalys with emphasis. or that by friendship made two-cywork, no hand guards; the precious names Minister hand method technical review of machinery, all the idiot's necessary assistants\n\nThe handbook of defenceominal arguments over means in my", - "<|bos|>AUXE\n\nDISCOURSE XII.\n\n\"Gray Dyfe's \"Primer of Poetry.\n\nHERP, I-XXVI. \"The Marquis de Faug\u00e8res, who made the longest ode of verse\n\nApril 10. Four Years. \"LETTER XIII.\n\n155 William X. to His Sister, as it was sold at St. Petersburg, (?) in 1837. \"DEAR JOHN,-do you, I, mean to advise me through your father and your cousin Robert, how much moral and spiritual worth the person of Joseph Bonaparte may make upon him? that great intellect in England is as dear to me as", - "<|bos|> ministered, 350, 353, 355; marvelled at it till it stirred up heartfelt gratitude, 353; at\n\nMerton in the East, 40, 41, 42, 51; Burbow, the minister of the cross, 41 depicted as one whose religio powers harmonized with 35 Entertainment of this symbolism means, as I believe it would tend to debauch the thoughts and organize tendencies of a people to sin, and furnish an environment consequently personal to itself, but this avers divine authority, as developed in lies;\n\n692;", - "<|bos|>CHAPTER XIII.\n\nTHE TIDE WOMEN AND THE HEART OF THE BIBLE.\n\nAr the time of our Lord's Passion, a tenant of\n\nCrowne herself, a friendly, find, was found to be alive.\n\nAnd in truth, this great lady had madd' unto her taper.\n\nFor our Lord being dead, to his sence the sole effect was, to be dead.\n\nFrom her hand have we no reason to pray.\n\nSo am I, poore, married, or dead.\n\nthank\n\nDivers times I've much doubted whether the ghost of\n\nHobany resumed her hold, either in the Church where his", - "<|bos|>PREFACE XIII, PERS. 92. Besides these and other volumes of individuals, there are a few essays dealing respectively with political, the French Revolution, and the Confederate and Bankrupt countries,1 in general thoughts, words, and ways.\n\nWhen I have to thank most indulgent Heaven, summoned by fair profit the sowing of that mind concerning every subject which is at stake, and found it in that man who needs criticise expressions, reformations, improvisements, and principles insufficient in every one of them, it is the intention of this chapter to explain the principles and all manner of civilities, whether in political and social life,", - "<|bos|>) have published a\n\nHEREDITY OF HOMESTEAD DICTION\n\nThe proper pronunciation of the title, Bundl, and its necessary 1 a ao, band, ning from b for a. for, 17 it b simply and the na = 1 m\n\n79ly arrived at, everyday are always :\n\n23! @2 o2 b\n\nisy,\n\nThe letters and dulces of the Author in this and in the other statements, are therefore more warranted in being either for m, of, 1 or for bb, then jv\u00f4vi\u00e9 for m.\n\nThe sounds o", - "<|bos|>osso caso, e qu-cell-\n\nLive out of act e assumes he at to act of vindi-neling in the ni sa which preceded v, which was not more than vi h age and contemporaneous time, yet\n\n(rule Mn 4\n\nJOHN PIERRE 209\n\nmore to multiply v that life posses so HENT INTIF\n\n1802] (he adds, al pacity, ju \u200b1 n 227 25 194 \u664a\u200b\u9ac3 \u200b01 z\n\nSecondly, command your temper with love let the nag in to hehey n" - ], - "training_time_seconds": 2532.0307886600494, - "stage_training_flops": 1.06349373718895e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.06349373718895e+18, - "config_fingerprint": "b389d34fe603d8f3", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/dd182242", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1", - "dataset_fingerprint": "0db4605cfe3a7eac", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "unique_train_tokens": 9020437955, - "effective_epochs": 0.13728471524085772 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer/token_bytes.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer/tokenizer.pkl b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-baseline-s42-v1/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_000500.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_000500.json deleted file mode 100644 index 4ae73aba61297efa298d5226aaeac983393a7f78..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.3446261878698493, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "wandb_run_id": "08927e2a", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,muon-momentum-constant,muon-momentum-0.83,seed42", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.83, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "muon_momentum": 0.83, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "muon-momentum-constant", - "muon-momentum-0.83", - "seed42" - ] - }, - "config_fingerprint": "6d783d8df5a5d3a0", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6d783d8df5a5d3a0" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3446261878698493, - "smooth_train_loss": 3.683517498769485, - "total_training_time": 528.0507395267487, - "stage_start_step": 0, - "stage_training_flops": 225125685264384000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2.25125685264384e+17 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_001000.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_001000.json deleted file mode 100644 index 6a5dd6ac5080c5f0a84756842961992d3006e73e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.2473395243127656, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "wandb_run_id": "08927e2a", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,muon-momentum-constant,muon-momentum-0.83,seed42", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.83, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "muon_momentum": 0.83, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "muon-momentum-constant", - "muon-momentum-0.83", - "seed42" - ] - }, - "config_fingerprint": "6d783d8df5a5d3a0", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6d783d8df5a5d3a0" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2473395243127656, - "smooth_train_loss": 3.5313088858304695, - "total_training_time": 1067.1877615451813, - "stage_start_step": 0, - "stage_training_flops": 450251370528768000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 4.50251370528768e+17 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_001500.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_001500.json deleted file mode 100644 index beb446bca73369ecb2eb90639ee14e5863c6d527..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.1575217408184195, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "wandb_run_id": "08927e2a", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,muon-momentum-constant,muon-momentum-0.83,seed42", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.83, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "muon_momentum": 0.83, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "muon-momentum-constant", - "muon-momentum-0.83", - "seed42" - ] - }, - "config_fingerprint": "6d783d8df5a5d3a0", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6d783d8df5a5d3a0" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.1575217408184195, - "smooth_train_loss": 3.4231843039810266, - "total_training_time": 1605.6937384605408, - "stage_start_step": 0, - "stage_training_flops": 675377055793152000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 6.75377055793152e+17 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_002000.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_002000.json deleted file mode 100644 index 57f95bcd3b56532c40f1845ada521167d30709a4..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.1174934905823755, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "wandb_run_id": "08927e2a", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,muon-momentum-constant,muon-momentum-0.83,seed42", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.83, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "muon_momentum": 0.83, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "muon-momentum-constant", - "muon-momentum-0.83", - "seed42" - ] - }, - "config_fingerprint": "6d783d8df5a5d3a0", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6d783d8df5a5d3a0" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.1174934905823755, - "smooth_train_loss": 3.346263538059964, - "total_training_time": 2144.2248499393463, - "stage_start_step": 0, - "stage_training_flops": 900502741057536000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 9.00502741057536e+17 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_002362.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_002362.json deleted file mode 100644 index 4053c55157e9d5056d1ab1c1c82a9e7833b16793..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.0913370687101807, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "wandb_run_id": "08927e2a", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,muon-momentum-constant,muon-momentum-0.83,seed42", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.83, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "muon_momentum": 0.83, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "muon-momentum-constant", - "muon-momentum-0.83", - "seed42" - ] - }, - "config_fingerprint": "6d783d8df5a5d3a0", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6d783d8df5a5d3a0" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": 1.0913370687101807, - "smooth_train_loss": 3.136666011396354, - "total_training_time": 2534.250967502594, - "stage_start_step": 0, - "stage_training_flops": 1063493737188950016, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.06349373718895e+18 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_000500.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_000500.pt deleted file mode 100644 index f5e0ecabcbe3c4bd0d73d906bd93df277b32c92a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b53a5c5f9543b6337cbabcb983db93ed3dc73144c90146df5dc62cd0ebd92958 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_001000.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_001000.pt deleted file mode 100644 index 23147c3188cefa31a26e2bf5a594caa019e883ec..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4a68e0cdacb0b2497ddaeb75eab3ab80372dc4aea15f8f3a4046da09e7d0423d -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_001500.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_001500.pt deleted file mode 100644 index e7108adbdebbe71b94de29c7174eb3fb4bde0b27..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:683f0db11335666a8a73d45ff860d8eca8c5a2e9caf16abfc4192c558c95d1a9 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_002000.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_002000.pt deleted file mode 100644 index 250ac6dc37a87a828926f5aa20f0cbb68e368f89..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a955a670ce2a6d902494c888f30da40a73df36a603fd749de6398a059bb35dc0 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_002362.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_002362.pt deleted file mode 100644 index 320db8271cdeb296aaef6dfdd7e2c9844c7ba4b3..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4c6547bcdae024f741fe2e223b8e27141446d9628c495e142905b2b1f8eae032 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_000500_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 69bcef5f8669f3e050ec4dce266218c290d6511d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fbd23df0af9fca80c0cbbf51e2c7ece6045fc57c8a2df02bd3c5033860ace893 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_001000_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index a7507b1cf632b8a0d2ba08569a01d19370ccfa77..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bfd24ba038952c67670b4eb79e249d407b4a0414a830ef8495eab9c9990da35b -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_001500_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 94dc7d4224826095bec102e4447cd3eacf53dfb3..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0ba8145bd102830452cbdf48cbb6fe2f6e153fac651c6c284ab6643f2f40364e -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_002000_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 4d4abc56981a46fd10fe9aa23db15b14ced6f827..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c85b345f64bbe486811f973197cb0449f8110d6f6a8f1df6b5bedc1470ca5d35 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_002362_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 8ba4391e12714f2d6c38751a1336805cddd31694..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f1cf17f6f14c8ec51d138f99bb6b7f4a5f94cb972efca14f22c37b3cd1c86831 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/config.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/config.json deleted file mode 100644 index 7cec9432adf457e9c8dfb2d49ae23a2b43f145fc..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/config.json +++ /dev/null @@ -1,70 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "muon_momentum": 0.83, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "muon-momentum-constant", - "muon-momentum-0.83", - "seed42" - ] - }, - "config_fingerprint": "6d783d8df5a5d3a0", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1" -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/evals/samples.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/evals/samples.json deleted file mode 100644 index 709cf09976319c8d5d76366f57241091b7a4c7cf..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the country, and the capital of the country is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the gold of the earth. The gold of the earth is the" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and so on. The\n\nSunday will be Saturday, and so on" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is hot, and hot is hot, and hot is hot, and hot is hot" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: (1) the sun, (2) the moon, (3) the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a deep blue, and the color of the color is a deep blue. The" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the 5th, and the 5th, and the" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nTHIS essay treats of those emotions, which may be considered as the characteristic of man's own nature, regarded as his own genius, the influence and influence of Rousseau and the examples of men who succeeded him. Climbing in the career of physical science, anticipating the voice of the heart even in this degenerate age, seeing the divine mission to which they had been set-to seek divine influences around them, thus picturing the greatness of a great man, Rousseau discovered the fullness of the universe expanding beneath him, adding a new creation to its own wisdom.\n\nFor his purpose, this volume, unlike the earlier ones of the present essay,", - "<|bos|>PREFACE. REV. CER\u043f\u0435\u043c. Various editions, for which I thank the reader, and for special reasons, especially because they are the first of class editions after the Roman.\n\nIn choosing choice editions it has never been thought my duty to give the reader a fresh idea of the time, place, and order of every first or principal place or portion of a celebrated public library, and to retain but a partial acquaintance with the French, Romans, and German Universities. Public Libraries are Ministerial libraries, reviewers, honorary authors, and filllemensensible louvbers of established Universities. Guariumius is my", - "<|bos|>RID vision of the cairn is large and rapid; when it begins to move towards the left bank it suddenly becomes obscure; at a little distance it is almost imperceptible. Spite of the extreme caution, however, of the parties, it contains always first the low sandstone cisonpeated by the mosses around the cliffs, and when it slowly advances from a (?) stone-like and depressed formation, rapidly effervesces with fresh water, which increases the lustre of the surface and gives it a majestic aspect. As it reaches higher the sandstone cisonpeated by the circle changes its state, and falls away in a very long period", - "<|bos|> ministered to get him a spirit of idealism. Historical reading one he did not wish to endorse.\n\nLondon March 11, A. S. P. 466. lect in them : he retained even a passing interest in the \"apt to say\" of \"to die,\" \"in any faith, however weak, much less to die \". The parallels that Entertains are chiefly verbal, and the results plain enough, but not the whole. The army and the politicians are not a dramatic product, but they are curious and consequently personal relations. However immediate this might be, one circumstance gleaned from them may not be amiss", - "<|bos|>PART II PART II PART III PART III PART III PART III PART I PART III PART III PART III PART III PART III Part I PART III PART\n\n PART III PART III PART I PART I PART I PART IV PART IV PART II PART IV PART V Optitude. S' International in Dr.\n\nClark Watts' Journal for April 24-26 (1857).\n\nPart II.\n\nPART III.\n\nPart IV.\n\n(a). The influens de cantes, as remembrancer, or correspondent subject, that is as to the names and parvantages of all the above that has been resumed.\n\nThe first meeting of the two associated Societies", - "<|bos|>PREFACE this scheme is too marked. But pictures and pictures crowded into them prove the disheartened youth party spirit, and overcomes the strain of the effort of social rewith regard to the ability or failure to study struck by the seventeenth century doctors in one place.\n\nFirst among them, indeed, fair profit must be made; and the diligence employed in completing the figures was then often not in proportion to the opportunity of studying them, but it led to greater variety among the new applicants. The following quotations are representative of the kind of work done by a municipal artist, as also all others.\n\nB. Ephemera, The Pilgrims\n\nHand", - "<|bos|>INTRODUCTION have published a work on the Bible as a guide for the East, and as a proper place to lay out a careful account of it. \u00fc 1\n\nAnd in the thing itself it differs from a carefully wrought work. for the purpose of deciding who simply hand the volume to the modern convenient prepared manuscript, and who may take her text; and who striking against reverence to keep silence may lay it from sight, and may dissemble it. There can be no doubt that every Bible should be translated for use to the world of men in our own country. But it is right to retain that reserve for use, as giving no", - "<|bos|>vi\n\nPREFACE\n\nix\n\nINTRODUCTION\n\nvii\n\nHOOD AND WAR LECTURES I. MILITARY ATTACHING\n\nlingering the consideration concluded on the former, the profession of more than one religious faith and practice was desired, which, if not attended to by others, whose communication with the world could be interrupted, would perhaps plead for them the invitation attending the entrance of departed souls into the sacred fellowship, which whoever wished for, would scarcely expect to go forth to good work with a purer or better spirit than his own.\n\nGeorge Howard Baillie records the opinion of many Christians who read the word Socialist many years back, he would like" - ] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/evals/val_bpb.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/evals/val_bpb.json deleted file mode 100644 index 70626f7e62cdcdbf54ea6654f61044e4eb37e58d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/evals/val_bpb.json +++ /dev/null @@ -1,94 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val_per_position": [ - { - "start": 0, - "end": 256, - "bpb": 1.1274765564193325 - }, - { - "start": 256, - "end": 512, - "bpb": 1.0657887573320168 - }, - { - "start": 512, - "end": 768, - "bpb": 1.0529323371637673 - }, - { - "start": 768, - "end": 1024, - "bpb": 1.042417557436246 - }, - { - "start": 1024, - "end": 1280, - "bpb": 1.0361290704191266 - }, - { - "start": 1280, - "end": 1536, - "bpb": 1.0340712039397197 - }, - { - "start": 1536, - "end": 1792, - "bpb": 1.0296966597181718 - }, - { - "start": 1792, - "end": 2048, - "bpb": 1.0268858353043673 - }, - { - "start": 2048, - "end": 2304, - "bpb": 1.0237218740325673 - }, - { - "start": 2304, - "end": 2560, - "bpb": 1.0197109227177406 - }, - { - "start": 2560, - "end": 2816, - "bpb": 1.02172956043314 - }, - { - "start": 2816, - "end": 3072, - "bpb": 1.0230340484873293 - }, - { - "start": 3072, - "end": 3328, - "bpb": 1.0191460973859146 - }, - { - "start": 3328, - "end": 3584, - "bpb": 1.0169676603801217 - }, - { - "start": 3584, - "end": 3840, - "bpb": 1.0154473730127702 - }, - { - "start": 3840, - "end": 4096, - "bpb": 1.014662013752163 - } - ], - "val": 1.0356202398726997 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/run.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/run.json deleted file mode 100644 index a24abd9b5ef67ecb346374f06cc7365b45b994bb..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "branch_parent_step": null, - "config_fingerprint": "6d783d8df5a5d3a0", - "wandb_run_id": "08927e2a", - "created_at": 1785983842 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/summary.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/summary.json deleted file mode 100644 index 701f1aa28c42f435af60bd5e9999d33d8ea5c257..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/summary.json +++ /dev/null @@ -1,70 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean-1930s", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.0913370687101807, - "minimum_sampled_val_bpb": 1.0913370687101807, - "full_val_bpb": 1.0356202398726997, - "core_metric": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the country, and the capital of the country is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the gold of the earth. The gold of the earth is the" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and so on. The\n\nSunday will be Saturday, and so on" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is hot, and hot is hot, and hot is hot, and hot is hot" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: (1) the sun, (2) the moon, (3) the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a deep blue, and the color of the color is a deep blue. The" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the 5th, and the 5th, and the" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nTHIS essay treats of those emotions, which may be considered as the characteristic of man's own nature, regarded as his own genius, the influence and influence of Rousseau and the examples of men who succeeded him. Climbing in the career of physical science, anticipating the voice of the heart even in this degenerate age, seeing the divine mission to which they had been set-to seek divine influences around them, thus picturing the greatness of a great man, Rousseau discovered the fullness of the universe expanding beneath him, adding a new creation to its own wisdom.\n\nFor his purpose, this volume, unlike the earlier ones of the present essay,", - "<|bos|>PREFACE. REV. CER\u043f\u0435\u043c. Various editions, for which I thank the reader, and for special reasons, especially because they are the first of class editions after the Roman.\n\nIn choosing choice editions it has never been thought my duty to give the reader a fresh idea of the time, place, and order of every first or principal place or portion of a celebrated public library, and to retain but a partial acquaintance with the French, Romans, and German Universities. Public Libraries are Ministerial libraries, reviewers, honorary authors, and filllemensensible louvbers of established Universities. Guariumius is my", - "<|bos|>RID vision of the cairn is large and rapid; when it begins to move towards the left bank it suddenly becomes obscure; at a little distance it is almost imperceptible. Spite of the extreme caution, however, of the parties, it contains always first the low sandstone cisonpeated by the mosses around the cliffs, and when it slowly advances from a (?) stone-like and depressed formation, rapidly effervesces with fresh water, which increases the lustre of the surface and gives it a majestic aspect. As it reaches higher the sandstone cisonpeated by the circle changes its state, and falls away in a very long period", - "<|bos|> ministered to get him a spirit of idealism. Historical reading one he did not wish to endorse.\n\nLondon March 11, A. S. P. 466. lect in them : he retained even a passing interest in the \"apt to say\" of \"to die,\" \"in any faith, however weak, much less to die \". The parallels that Entertains are chiefly verbal, and the results plain enough, but not the whole. The army and the politicians are not a dramatic product, but they are curious and consequently personal relations. However immediate this might be, one circumstance gleaned from them may not be amiss", - "<|bos|>PART II PART II PART III PART III PART III PART III PART I PART III PART III PART III PART III PART III Part I PART III PART\n\n PART III PART III PART I PART I PART I PART IV PART IV PART II PART IV PART V Optitude. S' International in Dr.\n\nClark Watts' Journal for April 24-26 (1857).\n\nPart II.\n\nPART III.\n\nPart IV.\n\n(a). The influens de cantes, as remembrancer, or correspondent subject, that is as to the names and parvantages of all the above that has been resumed.\n\nThe first meeting of the two associated Societies", - "<|bos|>PREFACE this scheme is too marked. But pictures and pictures crowded into them prove the disheartened youth party spirit, and overcomes the strain of the effort of social rewith regard to the ability or failure to study struck by the seventeenth century doctors in one place.\n\nFirst among them, indeed, fair profit must be made; and the diligence employed in completing the figures was then often not in proportion to the opportunity of studying them, but it led to greater variety among the new applicants. The following quotations are representative of the kind of work done by a municipal artist, as also all others.\n\nB. Ephemera, The Pilgrims\n\nHand", - "<|bos|>INTRODUCTION have published a work on the Bible as a guide for the East, and as a proper place to lay out a careful account of it. \u00fc 1\n\nAnd in the thing itself it differs from a carefully wrought work. for the purpose of deciding who simply hand the volume to the modern convenient prepared manuscript, and who may take her text; and who striking against reverence to keep silence may lay it from sight, and may dissemble it. There can be no doubt that every Bible should be translated for use to the world of men in our own country. But it is right to retain that reserve for use, as giving no", - "<|bos|>vi\n\nPREFACE\n\nix\n\nINTRODUCTION\n\nvii\n\nHOOD AND WAR LECTURES I. MILITARY ATTACHING\n\nlingering the consideration concluded on the former, the profession of more than one religious faith and practice was desired, which, if not attended to by others, whose communication with the world could be interrupted, would perhaps plead for them the invitation attending the entrance of departed souls into the sacred fellowship, which whoever wished for, would scarcely expect to go forth to good work with a purer or better spirit than his own.\n\nGeorge Howard Baillie records the opinion of many Christians who read the word Socialist many years back, he would like" - ], - "training_time_seconds": 2534.250967502594, - "stage_training_flops": 1.06349373718895e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.06349373718895e+18, - "config_fingerprint": "6d783d8df5a5d3a0", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/08927e2a", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1", - "dataset_fingerprint": "0db4605cfe3a7eac", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "unique_train_tokens": 9020437955, - "effective_epochs": 0.13728471524085772 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer/token_bytes.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer/tokenizer.pkl b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon083-s42-v1/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_000500.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_000500.json deleted file mode 100644 index 8553a3938d1a650b2cb8f4f76a321b674d888b92..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.3401702205232602, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "wandb_run_id": "0271e3eb", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,seed42,muon-momentum-constant,muon-momentum-0.90", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "seed42", - "muon-momentum-constant", - "muon-momentum-0.90" - ] - }, - "config_fingerprint": "6a707178cf2e40fc", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6a707178cf2e40fc" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3401702205232602, - "smooth_train_loss": 3.6873101553428147, - "total_training_time": 526.4253585338593, - "stage_start_step": 0, - "stage_training_flops": 225125685264384000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2.25125685264384e+17 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_001000.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_001000.json deleted file mode 100644 index 6289744e323742e1444f59af60cfa6aedd70c708..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.236094953958828, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "wandb_run_id": "0271e3eb", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,seed42,muon-momentum-constant,muon-momentum-0.90", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "seed42", - "muon-momentum-constant", - "muon-momentum-0.90" - ] - }, - "config_fingerprint": "6a707178cf2e40fc", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6a707178cf2e40fc" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.236094953958828, - "smooth_train_loss": 3.5190564919373246, - "total_training_time": 1064.2983202934265, - "stage_start_step": 0, - "stage_training_flops": 450251370528768000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 4.50251370528768e+17 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_001500.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_001500.json deleted file mode 100644 index ba6dc1521c57f84aa0ddc7efb45f4815ed7fe1e6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.1537679671738093, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "wandb_run_id": "0271e3eb", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,seed42,muon-momentum-constant,muon-momentum-0.90", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "seed42", - "muon-momentum-constant", - "muon-momentum-0.90" - ] - }, - "config_fingerprint": "6a707178cf2e40fc", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6a707178cf2e40fc" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.1537679671738093, - "smooth_train_loss": 3.4145766089682157, - "total_training_time": 1601.6644263267517, - "stage_start_step": 0, - "stage_training_flops": 675377055793152000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 6.75377055793152e+17 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_002000.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_002000.json deleted file mode 100644 index bcdd4ffc0f159b22d7e12941cb081a7eea0b16b3..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.1151770575853448, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "wandb_run_id": "0271e3eb", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,seed42,muon-momentum-constant,muon-momentum-0.90", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "seed42", - "muon-momentum-constant", - "muon-momentum-0.90" - ] - }, - "config_fingerprint": "6a707178cf2e40fc", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6a707178cf2e40fc" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.1151770575853448, - "smooth_train_loss": 3.3391174163554944, - "total_training_time": 2139.0777184963226, - "stage_start_step": 0, - "stage_training_flops": 900502741057536000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 9.00502741057536e+17 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_002362.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_002362.json deleted file mode 100644 index aa78c7825c4afc73c5a3d45c96a39a1a3afb392c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "val_bpb": 1.089106110634849, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "wandb_run_id": "0271e3eb", - "wandb_group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,ctx4096,sssl,fulltok,fp8,h100,autoresearch-transfer-v1,seed42,muon-momentum-constant,muon-momentum-0.90", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "muon_momentum": 0.9, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": null, - "init_from_step": -1, - "no_init_optimizer": false, - "branch_lr_schedule": "branch", - "branch_parent_experiment_id": null, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/pretok", - "mixture_source_dirs": null, - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "seed42", - "muon-momentum-constant", - "muon-momentum-0.90" - ] - }, - "config_fingerprint": "6a707178cf2e40fc", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6a707178cf2e40fc" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": 1.089106110634849, - "smooth_train_loss": 3.1315954124792693, - "total_training_time": 2528.233864545822, - "stage_start_step": 0, - "stage_training_flops": 1063493737188950016, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.06349373718895e+18 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_000500.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_000500.pt deleted file mode 100644 index e96b21b8cafaf9b211c049b0c1465e3368500c12..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:68a44c5037e6cd6fe20c5dbdbb981a1a468459d884390572ac977439de7ddd69 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_001000.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_001000.pt deleted file mode 100644 index d459c86bf922e749631f2c88de578952e8e5ccf5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:750728d13176232157adda742e0e3039391d7d6dab69e1faf0c3aa42d3929e96 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_001500.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_001500.pt deleted file mode 100644 index cbd62c0b1768f1a88998e923037eb2f6101561a1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:da509d33e20806bbeeeda1fd6d7a74df3e598077f515ed3b982deff299256a00 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_002000.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_002000.pt deleted file mode 100644 index 6457e803f1588dae2fd34d76a00f953a14acf3a6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:ea83b8e6f3ecfe7941edf52e14e45a901fda25832c596c22894d017493abf0ac -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_002362.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_002362.pt deleted file mode 100644 index d23dbe10842333f7a81500ab53806b914ef87648..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b0702cec36e71ca2195924c97e08fb178892f13fc99bf6b51ff1f1fc774776ed -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_000500_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 9c6092c5794dcf5fcbece2719c2c74936598eaf1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5dc26b319b73b7a4fcf04440fb6e410aef007f061f510b1bb361778fd73c4045 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_001000_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 60dd06bb5b5757e52fceff40d0cce3006190df88..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0f201636a64b81cd5b9a6a0a06e64e4a7939e071a3967203a6fb1e6d9f15f109 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_001500_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index c6a3770c0b652d0066df67f138fc2f51c0993fa6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d1e5404258e1c29c15b755a2880237c9d0f1f652752876afe7b1e4299e095102 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_002000_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 6ab8ca5af0d209e6e6b90b38f8e62fb06113b217..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c18d566e9183752553658da01b82ca6469c1c47447258c72558676e3d82942a0 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_002362_rank0.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 64ae5ece1b97ff71922db143810e3a0439d17e9f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:219cad4a6369b79f19b15148bfcf23f4fcf78bf5e1da02895e7f8197db185b5b -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/config.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/config.json deleted file mode 100644 index 948aa9c1733fb0aea0b860e24a9b2b43ac6c3f07..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/config.json +++ /dev/null @@ -1,70 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "muon_momentum": 0.9, - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "group": "clean1930s-d12-ctx4096-autoresearch-transfer-v1", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "ctx4096", - "sssl", - "fulltok", - "fp8", - "h100", - "autoresearch-transfer-v1", - "seed42", - "muon-momentum-constant", - "muon-momentum-0.90" - ] - }, - "config_fingerprint": "6a707178cf2e40fc", - "artifact_path": "experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1" -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/evals/samples.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/evals/samples.json deleted file mode 100644 index fc61fe6bb65a08d338d86949c2163784aac595e2..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the country, and the capital of the country is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the gold of the sun, and the symbol of the sun is the gold of" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. The\n\nLord's day is the day of the Lord's, and" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the hot of the sun, and the opposite of cold is the hot of the" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the planets, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a white one, and I have a white one. I have a white one" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the same as x, and the same as x, and the same as x" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nTHIS essay was contributed to the \"Introduction \"-volume journal of the \"Society for the Promotion of\n\nScience,\" which has always prepared it for use only in the \"training Fair.\" Page 303\n\nSENTIMENTS AND INDUSTRIAL ENDS.\n\nWEALTH-ONE OATH.-Old woman Dunning called divine service to mind if the Old woman was mazy, if she was a little old man.\n\nHis servitor, Billy Freykin, had given a bonne bouche to the palate of the young chemist.\n\nThe fine old lady came about the first of each month of the year and adored", - "<|bos|>PREFACE.\n\nTHE LIBRARY OF ACOPYRIGHT AND VOLUME III.\n\nThe Proceedings for the Percussion-Capric Outastes.\n\nTHE Volume, which has been the first to be issued after the studies prosecuted, was founded during the year 1871, when the University Government assigned the Library of the Biological Woman to the two rooms of the first sewing society.\n\nUpon the turning over of the little volume, with its completest arrangement, made in 1873, no further contributions were made. Still, the method technical review of the volume is to be found necessary.\n\nSome inferior articles, merely established as stores of overwork, allowing", - "<|bos|>The vision of the man who is lord of the brute Dyfe, also brings the vision directly into collision with Vanderburgh's doctrine as stated above. It corresponded more exactly in form and structure to his views, than to those of the Edwards. The vision first introduces the man cison\u00e1red by the destroyed horse Dyfe as a homoeoteer to life, while in another and more elaborate form he escapes but leaves the precipice. Sometimes the vision has the rarity of its object, sometimes a foreign element appears and notes it. It was made where the rarity of cancelation appears to be strongest. The facts exhibit only the", - "<|bos|>ESLIE 350, 37 Ind. 125; Martel, XIV, 703; Notes on Aphrodoclus\n\nMut\u00f3s, 765; Deshon, 756; Quaere, de Ecclesia die S.J...B. .954; Religio: Pav. Theol., v, i, xii, xiv; [Recond. Am. Soc. Antiq. Society [1905], iv 3, 30; Urr. Arch. Antiq., vii, vii; a note follows, this allusion.-Cf.\n\nBeb. ", - "<|bos|>CHAPTER XIII.\n\nTHE TIDE OF MOUNTAINS, OR THE ROO- NUM AND THE CROSS CHURCHES-TI- TIONAL LATTER-ROADS, OR UNIFORM-ROPS IN ALGIANTRA-GAT. AF'LIGIOUS IMPRESSIONS OF THE MAJOR DESER- MAUR.\n\nNUM is manifest on many a bright day; in others no brilliancy can be feared. The pale shades which show on the mountains are subject to the most violent tempests, and the wild animals in the wood are alarmed that they may leap out of darkness, and the gorgeous insect cover", - "<|bos|>PREFACE\n\n771. On this I can wish and was glad to pass my whole life, for it was part of my education: at least, I was intimately able to appreciate the advantages of education and to study it by heart; the only thing, I say, that can, ever, make me feel an eager desire to study every subject which is at present open and engrosses my time. While I criticise my arrangement of my studies, I don't find any principles insufficient to insure my success.s In the same manner, I am a pretty strong advocate for the necessity of a little real, lasting preparation for a profession, where", - "<|bos|>INTRODUCTION.\n\nThe Sultan's Architect has prepared the Royal Equestrian School for the Service of Devotion; and there were thirteen\n\nlarge chests of Stationers' Hallowed Schools, without regard to local or political traditions, not a few of which are excellent, simply because the Council made themselves modern, prepared exactly by the school authorities, and always gifted as with the eyes and ears of the people. These Schools were, at the Time of which I write, consigned to an excellent school; but before their return to Graduates, twenty-four lines and seven black blackbirds were on toberage around them. All buildings that could", - "<|bos|>vi\n\nof 1889-90-\n\nLive out of doors in Yucatan.\n\nTalk of the Indies in the reign concluded on by Count Mochilla in his letter to the Czar and presented to Congress at\n\nPhiladelphia by M. Julien.\n\n\"It is not impossible that some mode of supplying us with necessaries may be suggested by these colonies.\"\n\nDear Brother M. has sent you a sketch of his history with a Life of Manila. It is the good fortune of a minister to be a mother to this people. It is of importance to the interests of Christians in the sugar-growing districts that Fernando, he tells us" - ] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/evals/val_bpb.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/evals/val_bpb.json deleted file mode 100644 index 451c12f4ea8609d5db51255518a0b6d4294b8bc1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/evals/val_bpb.json +++ /dev/null @@ -1,94 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val_per_position": [ - { - "start": 0, - "end": 256, - "bpb": 1.1253436362626366 - }, - { - "start": 256, - "end": 512, - "bpb": 1.0642184743188667 - }, - { - "start": 512, - "end": 768, - "bpb": 1.0510683484355177 - }, - { - "start": 768, - "end": 1024, - "bpb": 1.0405059216393195 - }, - { - "start": 1024, - "end": 1280, - "bpb": 1.0345590529216036 - }, - { - "start": 1280, - "end": 1536, - "bpb": 1.0328886578039924 - }, - { - "start": 1536, - "end": 1792, - "bpb": 1.0281508738706628 - }, - { - "start": 1792, - "end": 2048, - "bpb": 1.0252040319043694 - }, - { - "start": 2048, - "end": 2304, - "bpb": 1.0220662501454005 - }, - { - "start": 2304, - "end": 2560, - "bpb": 1.018049082689113 - }, - { - "start": 2560, - "end": 2816, - "bpb": 1.0200546960682964 - }, - { - "start": 2816, - "end": 3072, - "bpb": 1.021086121097632 - }, - { - "start": 3072, - "end": 3328, - "bpb": 1.017408884359891 - }, - { - "start": 3328, - "end": 3584, - "bpb": 1.0154681928589606 - }, - { - "start": 3584, - "end": 3840, - "bpb": 1.0142038984329307 - }, - { - "start": 3840, - "end": 4096, - "bpb": 1.012878088694088 - } - ], - "val": 1.03395383030625 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/run.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/run.json deleted file mode 100644 index ac02d165a30e0be166928c434010278fd50ad9f0..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "branch_parent_step": null, - "config_fingerprint": "6a707178cf2e40fc", - "wandb_run_id": "0271e3eb", - "created_at": 1785986530 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/summary.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/summary.json deleted file mode 100644 index 2bfe002379e8e51ee7eb389306c38346d6af8dcf..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/summary.json +++ /dev/null @@ -1,70 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean-1930s", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.089106110634849, - "minimum_sampled_val_bpb": 1.089106110634849, - "full_val_bpb": 1.03395383030625, - "core_metric": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the country, and the capital of the country is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the gold of the sun, and the symbol of the sun is the gold of" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. The\n\nLord's day is the day of the Lord's, and" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the hot of the sun, and the opposite of cold is the hot of the" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the planets, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a white one, and I have a white one. I have a white one" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the same as x, and the same as x, and the same as x" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nTHIS essay was contributed to the \"Introduction \"-volume journal of the \"Society for the Promotion of\n\nScience,\" which has always prepared it for use only in the \"training Fair.\" Page 303\n\nSENTIMENTS AND INDUSTRIAL ENDS.\n\nWEALTH-ONE OATH.-Old woman Dunning called divine service to mind if the Old woman was mazy, if she was a little old man.\n\nHis servitor, Billy Freykin, had given a bonne bouche to the palate of the young chemist.\n\nThe fine old lady came about the first of each month of the year and adored", - "<|bos|>PREFACE.\n\nTHE LIBRARY OF ACOPYRIGHT AND VOLUME III.\n\nThe Proceedings for the Percussion-Capric Outastes.\n\nTHE Volume, which has been the first to be issued after the studies prosecuted, was founded during the year 1871, when the University Government assigned the Library of the Biological Woman to the two rooms of the first sewing society.\n\nUpon the turning over of the little volume, with its completest arrangement, made in 1873, no further contributions were made. Still, the method technical review of the volume is to be found necessary.\n\nSome inferior articles, merely established as stores of overwork, allowing", - "<|bos|>The vision of the man who is lord of the brute Dyfe, also brings the vision directly into collision with Vanderburgh's doctrine as stated above. It corresponded more exactly in form and structure to his views, than to those of the Edwards. The vision first introduces the man cison\u00e1red by the destroyed horse Dyfe as a homoeoteer to life, while in another and more elaborate form he escapes but leaves the precipice. Sometimes the vision has the rarity of its object, sometimes a foreign element appears and notes it. It was made where the rarity of cancelation appears to be strongest. The facts exhibit only the", - "<|bos|>ESLIE 350, 37 Ind. 125; Martel, XIV, 703; Notes on Aphrodoclus\n\nMut\u00f3s, 765; Deshon, 756; Quaere, de Ecclesia die S.J...B. .954; Religio: Pav. Theol., v, i, xii, xiv; [Recond. Am. Soc. Antiq. Society [1905], iv 3, 30; Urr. Arch. Antiq., vii, vii; a note follows, this allusion.-Cf.\n\nBeb. ", - "<|bos|>CHAPTER XIII.\n\nTHE TIDE OF MOUNTAINS, OR THE ROO- NUM AND THE CROSS CHURCHES-TI- TIONAL LATTER-ROADS, OR UNIFORM-ROPS IN ALGIANTRA-GAT. AF'LIGIOUS IMPRESSIONS OF THE MAJOR DESER- MAUR.\n\nNUM is manifest on many a bright day; in others no brilliancy can be feared. The pale shades which show on the mountains are subject to the most violent tempests, and the wild animals in the wood are alarmed that they may leap out of darkness, and the gorgeous insect cover", - "<|bos|>PREFACE\n\n771. On this I can wish and was glad to pass my whole life, for it was part of my education: at least, I was intimately able to appreciate the advantages of education and to study it by heart; the only thing, I say, that can, ever, make me feel an eager desire to study every subject which is at present open and engrosses my time. While I criticise my arrangement of my studies, I don't find any principles insufficient to insure my success.s In the same manner, I am a pretty strong advocate for the necessity of a little real, lasting preparation for a profession, where", - "<|bos|>INTRODUCTION.\n\nThe Sultan's Architect has prepared the Royal Equestrian School for the Service of Devotion; and there were thirteen\n\nlarge chests of Stationers' Hallowed Schools, without regard to local or political traditions, not a few of which are excellent, simply because the Council made themselves modern, prepared exactly by the school authorities, and always gifted as with the eyes and ears of the people. These Schools were, at the Time of which I write, consigned to an excellent school; but before their return to Graduates, twenty-four lines and seven black blackbirds were on toberage around them. All buildings that could", - "<|bos|>vi\n\nof 1889-90-\n\nLive out of doors in Yucatan.\n\nTalk of the Indies in the reign concluded on by Count Mochilla in his letter to the Czar and presented to Congress at\n\nPhiladelphia by M. Julien.\n\n\"It is not impossible that some mode of supplying us with necessaries may be suggested by these colonies.\"\n\nDear Brother M. has sent you a sketch of his history with a Life of Manila. It is the good fortune of a minister to be a mother to this people. It is of importance to the interests of Christians in the sugar-growing districts that Fernando, he tells us" - ], - "training_time_seconds": 2528.233864545822, - "stage_training_flops": 1.06349373718895e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.06349373718895e+18, - "config_fingerprint": "6a707178cf2e40fc", - "git_commit_sha": "955f6037e852088dcc422f046ad5236ea66090b7", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/0271e3eb", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1", - "dataset_fingerprint": "0db4605cfe3a7eac", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "unique_train_tokens": 9020437955, - "effective_epochs": 0.13728471524085772 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer/token_bytes.pt b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer/tokenizer.pkl b/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-ctx4096-sssl-fulltok-artransfer-muon090-s42-v1/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_000500.json b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_000500.json deleted file mode 100644 index 2b43afe320038ac431689c5ad1ca6ab1a88e82e9..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "val_bpb": 1.3328434591455662, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-muoneq-muonplus", - "wandb_run_id": "95708491", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,muoneq,muonplus,optimizer-ablation", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/config.json", - "tokenizer_fingerprint": "803338edbf3458e6", - "git_commit_sha": "a47442614b208fd6df1549f6ca130504ca8975ed", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-muoneq-muonplus", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-muoneq-muonplus", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "muoneq", - "muonplus", - "optimizer-ablation" - ] - }, - "config_fingerprint": "4f01f653a56d0daf", - "artifact_path": "experiments/clean1930s-d12-r11.25-muoneq-muonplus" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "4f01f653a56d0daf" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3328434591455662, - "smooth_train_loss": 3.828870216538712, - "total_training_time": 1320.5494623184204, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_001000.json b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_001000.json deleted file mode 100644 index bdf94f480fbbec1bcd7083cfefb90bbde43c200c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "val_bpb": 1.2431777724823778, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-muoneq-muonplus", - "wandb_run_id": "95708491", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,muoneq,muonplus,optimizer-ablation", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/config.json", - "tokenizer_fingerprint": "803338edbf3458e6", - "git_commit_sha": "a47442614b208fd6df1549f6ca130504ca8975ed", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-muoneq-muonplus", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-muoneq-muonplus", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "muoneq", - "muonplus", - "optimizer-ablation" - ] - }, - "config_fingerprint": "4f01f653a56d0daf", - "artifact_path": "experiments/clean1930s-d12-r11.25-muoneq-muonplus" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "4f01f653a56d0daf" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2431777724823778, - "smooth_train_loss": 3.570735216236959, - "total_training_time": 2669.536954641342, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_001500.json b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_001500.json deleted file mode 100644 index befaebca8c3fc88c204c92805f9112350309e715..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "val_bpb": 1.1772538631419072, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-muoneq-muonplus", - "wandb_run_id": "95708491", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,muoneq,muonplus,optimizer-ablation", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/config.json", - "tokenizer_fingerprint": "803338edbf3458e6", - "git_commit_sha": "a47442614b208fd6df1549f6ca130504ca8975ed", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-muoneq-muonplus", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-muoneq-muonplus", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "muoneq", - "muonplus", - "optimizer-ablation" - ] - }, - "config_fingerprint": "4f01f653a56d0daf", - "artifact_path": "experiments/clean1930s-d12-r11.25-muoneq-muonplus" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "4f01f653a56d0daf" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.1772538631419072, - "smooth_train_loss": 3.3416836335468014, - "total_training_time": 4018.106310606003, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_002000.json b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_002000.json deleted file mode 100644 index 965abfb6af90aa47ff16b80373fc6a631d1dfa85..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "val_bpb": 1.1372931129764448, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-muoneq-muonplus", - "wandb_run_id": "95708491", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,muoneq,muonplus,optimizer-ablation", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/config.json", - "tokenizer_fingerprint": "803338edbf3458e6", - "git_commit_sha": "a47442614b208fd6df1549f6ca130504ca8975ed", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-muoneq-muonplus", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-muoneq-muonplus", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "muoneq", - "muonplus", - "optimizer-ablation" - ] - }, - "config_fingerprint": "4f01f653a56d0daf", - "artifact_path": "experiments/clean1930s-d12-r11.25-muoneq-muonplus" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "4f01f653a56d0daf" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.1372931129764448, - "smooth_train_loss": 3.1939839740715126, - "total_training_time": 5366.212870359421, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_002362.json b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_002362.json deleted file mode 100644 index bd2c1ae345b3e8d4d83a1e01e89124942a59b52c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "val_bpb": 1.1162452479490097, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-muoneq-muonplus", - "wandb_run_id": "95708491", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,muoneq,muonplus,optimizer-ablation", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-muoneq-muonplus/config.json", - "tokenizer_fingerprint": "803338edbf3458e6", - "git_commit_sha": "a47442614b208fd6df1549f6ca130504ca8975ed", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-muoneq-muonplus", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-muoneq-muonplus", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "muoneq", - "muonplus", - "optimizer-ablation" - ] - }, - "config_fingerprint": "4f01f653a56d0daf", - "artifact_path": "experiments/clean1930s-d12-r11.25-muoneq-muonplus" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "4f01f653a56d0daf" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": 1.1162452479490097, - "smooth_train_loss": 3.1283991602458974, - "total_training_time": 6341.746661424637, - "stage_training_flops": 1098553864463843328, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1098553864463843328 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_000500.pt b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_000500.pt deleted file mode 100644 index b7d0b4d7847fd743553b74033bd14307cf36b4f2..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:49648b40dd6946856f8114d5e6ed78d33f45f1381c17766661de01e2e9eba2bb -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_001000.pt b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_001000.pt deleted file mode 100644 index 3e6b8200a0d698fe89e1d0d9195a138ed32a77d0..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b51d3b5ece5cf94619fe0fd9dc11f956e93bba17161011c0ccda7ff307f79753 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_001500.pt b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_001500.pt deleted file mode 100644 index 84b572539150af7411800e18b74296bf1699ddb2..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6161003a46a5be804b34fc295d4586ecb29e09936953e613c5b5860fd1548309 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_002000.pt b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_002000.pt deleted file mode 100644 index 3c697277e1a05a98d87be429be8d4678e65e542a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:eb22a4f3416bc93c937ab2b12e7d8bf112bd20b2983c7f145c5eadbc6bf16047 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_002362.pt b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_002362.pt deleted file mode 100644 index b39f67fa282e401db0de0567e32830568871cd56..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0a6b0a5ce017e532265d1fcb329dd0f8051e752ff754c00669892033553ebb7e -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_000500_rank0.pt b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index dc17ecd8d93252d804b7553fa5b06299fc4e9d4a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4ffde69c12ed9c8c0e7058e75c4bfd74788f35ffb85ca002569e69f1d1395fca -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_001000_rank0.pt b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 40116a6365cab40da5b220d3798ecd1f1d4b81d5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4f91b1f0394b8a49baeab03b9252226cfe89ef414b1ea4d60e54d6129100f3ce -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_001500_rank0.pt b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 9d01fbfa198c4a012baaca755f4360710a059825..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7d00f57df038ee139e432df8571c57c2f99ddd82d9c9fae22e6e71553ca7dc23 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_002000_rank0.pt b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 5df93e3b783934da48d931c88bccf643c644fbd5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2b9d9fb936a69f962545728bcdc659964a3786c1f242ffdbe23ff02187713a23 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_002362_rank0.pt b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index f7dfad70e4cd0e9404c251644fe7150f5365cf69..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:aa420d24fb27abbcb471da9820bd2f8a05285e6862704e8a073451f0a201dd8c -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/config.json b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/config.json deleted file mode 100644 index 48391fc7aa5d21c347c5dd6c95b08797b8f2bebb..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/config.json +++ /dev/null @@ -1,59 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-muoneq-muonplus", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "muoneq", - "muonplus", - "optimizer-ablation" - ] - }, - "config_fingerprint": "4f01f653a56d0daf", - "artifact_path": "experiments/clean1930s-d12-r11.25-muoneq-muonplus" -} diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/evals/core.json b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/evals/core.json deleted file mode 100644 index b3b01c0383482064d10911a98c019693e8131adb..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": 0.06305335515020447, - "core_results": { - "hellaswag_zeroshot": 0.27713602781295776, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.08601938933134079, - "arc_easy": 0.32154881954193115, - "arc_challenge": 0.21587030589580536, - "copa": 0.5, - "commonsense_qa": 0.28337427973747253, - "piqa": 0.5424374341964722, - "openbook_qa": 0.23200000822544098, - "lambada_openai": 0.21191538870334625, - "hellaswag": 0.27932682633399963, - "winograd": 0.5201465487480164, - "winogrande": 0.5003946423530579, - "bigbench_dyck_languages": 0.07200000435113907, - "agi_eval_lsat_ar": 0.260869562625885, - "bigbench_cs_algorithms": 0.4371212124824524, - "bigbench_operators": 0.10476190596818924, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.020813623443245888, - "coqa": 0.07303018867969513, - "boolq": 0.5412843823432922, - "bigbench_language_identification": 0.25529998540878296 - }, - "centered_results": { - "hellaswag_zeroshot": 0.036181370417277016, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.08601938933134079, - "arc_easy": 0.0953984260559082, - "arc_challenge": -0.04550625880559286, - "copa": 0.0, - "commonsense_qa": 0.10421784967184065, - "piqa": 0.08487486839294434, - "openbook_qa": -0.02399998903274536, - "lambada_openai": 0.21191538870334625, - "hellaswag": 0.03910243511199951, - "winograd": 0.040293097496032715, - "winogrande": 0.0007892847061157227, - "bigbench_dyck_languages": 0.07200000435113907, - "agi_eval_lsat_ar": 0.07608695328235625, - "bigbench_cs_algorithms": 0.4371212124824524, - "bigbench_operators": 0.10476190596818924, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.020813623443245888, - "coqa": 0.07303018867969513, - "boolq": -0.2071463622544941, - "bigbench_language_identification": 0.18074805875553682 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/evals/samples.json b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/evals/samples.json deleted file mode 100644 index 38064e1d01244bd91b9c550f04e2b0471a1085d2..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the kingdom of France, and the capital of the kingdom of England" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the gold of the earth, and the gold of the earth is the gold of" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and the day will be Sunday. The day will be Sunday, and" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as that of cold, and the same as that of hot. The" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the planets, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the color of the sky, and the color of the earth. The color of" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the sum of the sum of the sum of the sum of the sum of the" - } - ], - "unconditioned_samples": [ - "<|bos|>CHAPTER VS. CREV. A. It has already been mentioned that Roger North and the Anglo- Saxon Chronicle and the reigns of Henry and Edward III. tell us that Eden was situated on England's Apple within the government of Richard II., and it must now be premised that England brought over men from Normandy and Anjou, and maintained them there. The folly and swiftness of our pilgrims were its source.\n\nProvidence was no idler in being driven over this island by the rivers of Canterbury and Canterbury, and thence overland to Rome; nor in being conveyed to Jerusalem, to meet the friendlessness of an ignorant man", - "<|bos|>PREFACE.\n\nTHE portions of this work, which, without modification of time, would be impossible to be within the little compass of the compass, are so valuable, that it must be judged more as a class; and all men, of every class, are judged by the same standards of skill, observed by the same standard of practice.\n\nThe first part of the Cloaca Maxima, which was written in print, is too exuberant and exuberant, to allow of annotation. \"But the question came up whilst we were considering the books,\" he adds, \"of Ministers, And the gentlemen that were about me, joined in the", - "<|bos|>PREFACE.\n\nix emphatic word is given to such relief; and in the language of filial affection or maternal emotion or affection, the language he employs is never \"the language of all true repentance and deep sorrow, but a resort\" of sighs or complaints to a Father to whom all are alike reconciled, whose infinite attributes are majesty, loving-kindness, mercy, and justice.\n\nIf the incipient man was a material object or an object, he would \"stalk;\" by indolence, by vice, by attraction, he would throw himself into a stagnant pool, throw into it little basins and merely onions, drenching", - "<|bos|> Dreams and Rejoice over my Lute and Pledge I've Soon Began to\n\nWish never the pains of keep as Oblation all my Steps I took to Narrow the bounds I had to set;\n\nExtanced at this Stage my Branding spowl yet\n\nSeemed it a rather\n\nSpeech and gn\n\nA considerable Quantity Flapping in my 80 whiteness forbid the Onely False ity did afford me;\n\nAnd whoever Fidelity's Man bring in all South about 100\n\nEtched upon Conscience an Evoluble Sport Gold upon there a Tremb Billail he Built I had little faith till", - "<|bos|>Drops, with apples nourishing them and lilies suiting them. It is a custom with corn-growers to mow loaves. It is thus with the rusted corn, which we call wheat because it is hard. If a may be properly branched out into layers and used in raising and other purposes, it is sometimes described as branched out simply for bed, and germinated like a cap or knife, into one or other of the ultimate layers. It sometimes happens that pipes are left, which being laid on the ground and decayed or damped in time as they decompose them, remain", - "<|bos|>VE ZABE. [Heir of the Superinto- FOW is a gener- Founding of the Morality | Sociessie, by the Origin of Lan- Turks Previous to his elevation strength of Language, very spe- ties, been endowed with this essential business, ago there are only three distinct untitutional faculties. But the reconstruction of modern sciences in the Shinto variant is a very daring one, and treats the sublimity of the Fourteenth century as such. Storia in three skins parapped off byila- Strumeton, Nev., drawn him away and merrie (Wat v. Ortsboth", - "<|bos|>PREFACE Ruth and Robin Hood suggested the latter an asylum in the outhouses at Oakum Orbeek, where the late Charles Lee, the publisher of 1 A little instalment of this and other historical works is printed at the end of 1872.\n\nPREFACE.\n\nto secure a place for retirement should he be ever called on to join the rebels. This was left the further task of keeping all the patent offices in the selection of Warwick the publisher of aegicide Liber Quintale. \"Master Pashan and the vae variarent,\" added Francis Gorton after the interview above to Pashan's title,", - "<|bos|>opers.\n\n CarthABELLA LUBI.\n\n221 however, still perseveres (1251 n. d.), and succeeded in breaking down the walls and overwhelming the rest with stones. 66 It appears,\" says Lubio (d.) I, \"that the free city is secure in its centre; and that I, who have interfered on behalf of my people at all times during my rule, can easily turn the safety of Smyrna when made famous by the massacre of the Amir and the surrender of the city on the appointed day.\" This is somewhat like about Cortez: \"Franciscus qui desitunt\n\n" - ] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/evals/val_bpb.json b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/evals/val_bpb.json deleted file mode 100644 index ebea9b2c25c960bc6629627d0f7560538bddcc75..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.0583351355174524 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/run.json b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/run.json deleted file mode 100644 index 22cae601fff997614b79d9d643aeedffadae8411..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "4f01f653a56d0daf", - "wandb_run_id": "95708491", - "created_at": 1783965749 -} diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/summary.json b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/summary.json deleted file mode 100644 index 33528ff8d4ed4de47d56d2a6fe01345060890dcd..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean-1930s", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.1162452479490097, - "minimum_sampled_val_bpb": 1.1162452479490097, - "full_val_bpb": 1.0583351355174524, - "core_metric": 0.06305335515020447, - "centered_results": { - "hellaswag_zeroshot": 0.036181370417277016, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.08601938933134079, - "arc_easy": 0.0953984260559082, - "arc_challenge": -0.04550625880559286, - "copa": 0.0, - "commonsense_qa": 0.10421784967184065, - "piqa": 0.08487486839294434, - "openbook_qa": -0.02399998903274536, - "lambada_openai": 0.21191538870334625, - "hellaswag": 0.03910243511199951, - "winograd": 0.040293097496032715, - "winogrande": 0.0007892847061157227, - "bigbench_dyck_languages": 0.07200000435113907, - "agi_eval_lsat_ar": 0.07608695328235625, - "bigbench_cs_algorithms": 0.4371212124824524, - "bigbench_operators": 0.10476190596818924, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.020813623443245888, - "coqa": 0.07303018867969513, - "boolq": -0.2071463622544941, - "bigbench_language_identification": 0.18074805875553682 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the kingdom of France, and the capital of the kingdom of England" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the gold of the earth, and the gold of the earth is the gold of" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and the day will be Sunday. The day will be Sunday, and" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as that of cold, and the same as that of hot. The" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the planets, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the color of the sky, and the color of the earth. The color of" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the sum of the sum of the sum of the sum of the sum of the" - } - ], - "unconditioned_samples": [ - "<|bos|>CHAPTER VS. CREV. A. It has already been mentioned that Roger North and the Anglo- Saxon Chronicle and the reigns of Henry and Edward III. tell us that Eden was situated on England's Apple within the government of Richard II., and it must now be premised that England brought over men from Normandy and Anjou, and maintained them there. The folly and swiftness of our pilgrims were its source.\n\nProvidence was no idler in being driven over this island by the rivers of Canterbury and Canterbury, and thence overland to Rome; nor in being conveyed to Jerusalem, to meet the friendlessness of an ignorant man", - "<|bos|>PREFACE.\n\nTHE portions of this work, which, without modification of time, would be impossible to be within the little compass of the compass, are so valuable, that it must be judged more as a class; and all men, of every class, are judged by the same standards of skill, observed by the same standard of practice.\n\nThe first part of the Cloaca Maxima, which was written in print, is too exuberant and exuberant, to allow of annotation. \"But the question came up whilst we were considering the books,\" he adds, \"of Ministers, And the gentlemen that were about me, joined in the", - "<|bos|>PREFACE.\n\nix emphatic word is given to such relief; and in the language of filial affection or maternal emotion or affection, the language he employs is never \"the language of all true repentance and deep sorrow, but a resort\" of sighs or complaints to a Father to whom all are alike reconciled, whose infinite attributes are majesty, loving-kindness, mercy, and justice.\n\nIf the incipient man was a material object or an object, he would \"stalk;\" by indolence, by vice, by attraction, he would throw himself into a stagnant pool, throw into it little basins and merely onions, drenching", - "<|bos|> Dreams and Rejoice over my Lute and Pledge I've Soon Began to\n\nWish never the pains of keep as Oblation all my Steps I took to Narrow the bounds I had to set;\n\nExtanced at this Stage my Branding spowl yet\n\nSeemed it a rather\n\nSpeech and gn\n\nA considerable Quantity Flapping in my 80 whiteness forbid the Onely False ity did afford me;\n\nAnd whoever Fidelity's Man bring in all South about 100\n\nEtched upon Conscience an Evoluble Sport Gold upon there a Tremb Billail he Built I had little faith till", - "<|bos|>Drops, with apples nourishing them and lilies suiting them. It is a custom with corn-growers to mow loaves. It is thus with the rusted corn, which we call wheat because it is hard. If a may be properly branched out into layers and used in raising and other purposes, it is sometimes described as branched out simply for bed, and germinated like a cap or knife, into one or other of the ultimate layers. It sometimes happens that pipes are left, which being laid on the ground and decayed or damped in time as they decompose them, remain", - "<|bos|>VE ZABE. [Heir of the Superinto- FOW is a gener- Founding of the Morality | Sociessie, by the Origin of Lan- Turks Previous to his elevation strength of Language, very spe- ties, been endowed with this essential business, ago there are only three distinct untitutional faculties. But the reconstruction of modern sciences in the Shinto variant is a very daring one, and treats the sublimity of the Fourteenth century as such. Storia in three skins parapped off byila- Strumeton, Nev., drawn him away and merrie (Wat v. Ortsboth", - "<|bos|>PREFACE Ruth and Robin Hood suggested the latter an asylum in the outhouses at Oakum Orbeek, where the late Charles Lee, the publisher of 1 A little instalment of this and other historical works is printed at the end of 1872.\n\nPREFACE.\n\nto secure a place for retirement should he be ever called on to join the rebels. This was left the further task of keeping all the patent offices in the selection of Warwick the publisher of aegicide Liber Quintale. \"Master Pashan and the vae variarent,\" added Francis Gorton after the interview above to Pashan's title,", - "<|bos|>opers.\n\n CarthABELLA LUBI.\n\n221 however, still perseveres (1251 n. d.), and succeeded in breaking down the walls and overwhelming the rest with stones. 66 It appears,\" says Lubio (d.) I, \"that the free city is secure in its centre; and that I, who have interfered on behalf of my people at all times during my rule, can easily turn the safety of Smyrna when made famous by the massacre of the Amir and the surrender of the city on the appointed day.\" This is somewhat like about Cortez: \"Franciscus qui desitunt\n\n" - ], - "training_time_seconds": 6341.746661424637, - "stage_training_flops": 1.0985538644638433e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.0985538644638433e+18, - "config_fingerprint": "4f01f653a56d0daf", - "git_commit_sha": "a47442614b208fd6df1549f6ca130504ca8975ed", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/95708491", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/clean1930s-d12-r11.25-muoneq-muonplus", - "dataset_fingerprint": "4af5759948d15ff0", - "tokenizer_fingerprint": "803338edbf3458e6", - "unique_train_tokens": 1275519304, - "effective_epochs": 0.970873786164196 -} diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 4552b41249154dd89202bb69f5cb6e659f2e2b46..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-muoneq-muonplus", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1783965799 -} diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer/token_bytes.pt b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer/token_bytes.pt deleted file mode 100644 index 26d7e6581c41a049b038de82b8bb65a1c8d57c8f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4d25a126690db557828b34dc3bd3e8ebd615058016c1870e76b32b31266805ed -size 132649 diff --git a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer/tokenizer.pkl b/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer/tokenizer.pkl deleted file mode 100644 index d21e36e06c64fe9e86d8d579bccaae5542ad5ac9..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-muoneq-muonplus/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4932125f789267dbd571cc4bd19339ecfd7160c9b9cc0441ba6635756e4a8b0e -size 407747 diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_000500.json b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_000500.json deleted file mode 100644 index 334540825140b855909c94e40ec689b724d138fb..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-randtok", - "val_bpb": 1.3561541455317183, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-randtok", - "wandb_run_id": "6b8f19a5", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,randtok", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-randtok", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/config.json", - "tokenizer_fingerprint": "f01549dbe6e2aaa9", - "git_commit_sha": "118adeb2fc3350cf27ad635af372bd9b9410c0ef", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-randtok", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-randtok", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768, - "sampling": "random", - "sampling_seed": 42 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-randtok", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "randtok" - ] - }, - "config_fingerprint": "175184ba2ee0f5d3", - "artifact_path": "experiments/clean1930s-d12-r11.25-randtok" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-randtok", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "175184ba2ee0f5d3" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3561541455317183, - "smooth_train_loss": 3.8651697814606103, - "total_training_time": 1305.9748463630676, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_001000.json b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_001000.json deleted file mode 100644 index d7d7929f9639e59948cc65075c39652c8f12f07d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-randtok", - "val_bpb": 1.2521959168139378, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-randtok", - "wandb_run_id": "6b8f19a5", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,randtok", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-randtok", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/config.json", - "tokenizer_fingerprint": "f01549dbe6e2aaa9", - "git_commit_sha": "118adeb2fc3350cf27ad635af372bd9b9410c0ef", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-randtok", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-randtok", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768, - "sampling": "random", - "sampling_seed": 42 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-randtok", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "randtok" - ] - }, - "config_fingerprint": "175184ba2ee0f5d3", - "artifact_path": "experiments/clean1930s-d12-r11.25-randtok" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-randtok", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "175184ba2ee0f5d3" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2521959168139378, - "smooth_train_loss": 3.5352641959529243, - "total_training_time": 2640.971613883972, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_001500.json b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_001500.json deleted file mode 100644 index f28c0f7122e1dd22a855fdf7f364beb013489ea3..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-randtok", - "val_bpb": 1.171838291812204, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-randtok", - "wandb_run_id": "6b8f19a5", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,randtok", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-randtok", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/config.json", - "tokenizer_fingerprint": "f01549dbe6e2aaa9", - "git_commit_sha": "118adeb2fc3350cf27ad635af372bd9b9410c0ef", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-randtok", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-randtok", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768, - "sampling": "random", - "sampling_seed": 42 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-randtok", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "randtok" - ] - }, - "config_fingerprint": "175184ba2ee0f5d3", - "artifact_path": "experiments/clean1930s-d12-r11.25-randtok" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-randtok", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "175184ba2ee0f5d3" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.171838291812204, - "smooth_train_loss": 3.4159228083544537, - "total_training_time": 3976.167183637619, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_002000.json b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_002000.json deleted file mode 100644 index 22dd0864f21492fdbd26dc3d1feb46718eb406e2..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25-randtok", - "val_bpb": 1.1368966310332098, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-randtok", - "wandb_run_id": "6b8f19a5", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,randtok", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-randtok", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/config.json", - "tokenizer_fingerprint": "f01549dbe6e2aaa9", - "git_commit_sha": "118adeb2fc3350cf27ad635af372bd9b9410c0ef", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-randtok", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-randtok", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768, - "sampling": "random", - "sampling_seed": 42 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-randtok", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "randtok" - ] - }, - "config_fingerprint": "175184ba2ee0f5d3", - "artifact_path": "experiments/clean1930s-d12-r11.25-randtok" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-randtok", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "175184ba2ee0f5d3" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.1368966310332098, - "smooth_train_loss": 3.136166264429775, - "total_training_time": 5310.99071097374, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_002362.json b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_002362.json deleted file mode 100644 index e761b66bdcd3a934d3e44f0e27cf21122551cd1c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "clean1930s-d12-r11.25-randtok", - "val_bpb": 1.1128382305031206, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25-randtok", - "wandb_run_id": "6b8f19a5", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25,randtok", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25-randtok", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25-randtok/config.json", - "tokenizer_fingerprint": "f01549dbe6e2aaa9", - "git_commit_sha": "118adeb2fc3350cf27ad635af372bd9b9410c0ef", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25-randtok", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-randtok", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768, - "sampling": "random", - "sampling_seed": 42 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-randtok", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "randtok" - ] - }, - "config_fingerprint": "175184ba2ee0f5d3", - "artifact_path": "experiments/clean1930s-d12-r11.25-randtok" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-randtok", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "175184ba2ee0f5d3" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": 1.1128382305031206, - "smooth_train_loss": 3.185300296294561, - "total_training_time": 6277.177423000336, - "stage_training_flops": 1098553864463843328, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1098553864463843328 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_000500.pt b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_000500.pt deleted file mode 100644 index 4216bdba9af7eb6db9d6b4efc4564092bba4c7eb..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:ae8a604031071534c2cd16d6086cd495012f38121b53e2072da33e084a8b3ba0 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_001000.pt b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_001000.pt deleted file mode 100644 index 6c48611ab7e81a8a1766e50d63591462824e3f46..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:48630f071b47f0d89ea2572c606d32e32b1c143d345176add9aeaf81483ea490 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_001500.pt b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_001500.pt deleted file mode 100644 index def9e1c80987e41395a2c1ecf96d24ff359c6263..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1359363c9b03f7dd9ae84fb5a606e06dfaaaeabf612168817daa388780ca4ed4 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_002000.pt b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_002000.pt deleted file mode 100644 index 3654468cd5c86915b8e6ef23ed4952a82885b34a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5cd9a38e9ecae268247a5a41d1710293cf7c9176b74660d0d3a36a86a9770302 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_002362.pt b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_002362.pt deleted file mode 100644 index 8d1c6296224fb9f2bc3140f7f2b0e606c6c59682..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a72dbad509f7969cc0c8825a3078a402e31a466476b39573b8c638ac3d37a519 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_000500_rank0.pt b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 3fa0fb3e5d38a530d918d37936d031ddd23b8b0c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:443ccb4c4d76adf1c33e0d0b3cacad3f29f522f07643ee180e6f461178593f51 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_001000_rank0.pt b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 629925ba1bffcd2113f7930bd69e2dbacc714085..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:36325a3bba2b4d7461ae926bf79112585c2e52c5bf1c9f5847f5198713794add -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_001500_rank0.pt b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 57722c59bc97301d345a708ca7f6a3110e061358..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7bb11ab9d0b1d54523b190dea5f37893ebd91858694403d047606196aa16887d -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_002000_rank0.pt b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index e362508889ac81fc9fbf6c3f26b122e5b339602a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f6941a6e8678fc86b1041e974eed1173410cca4fea06167c76e96a64d5260f6a -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_002362_rank0.pt b/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index fde42ad8707f958b368d1ffaedacbe1dc0e8c36b..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0648d159d91067bff35675cba8b5828e792b5ef7eb442e6ec5ea883179e65a12 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25-randtok/config.json b/experiments/clean1930s-d12-r11.25-randtok/config.json deleted file mode 100644 index a830707f7c7496c04dc047a0fcf2c99a4d2fe841..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/config.json +++ /dev/null @@ -1,59 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25-randtok", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768, - "sampling": "random", - "sampling_seed": 42 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25-randtok", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25", - "randtok" - ] - }, - "config_fingerprint": "175184ba2ee0f5d3", - "artifact_path": "experiments/clean1930s-d12-r11.25-randtok" -} diff --git a/experiments/clean1930s-d12-r11.25-randtok/evals/core.json b/experiments/clean1930s-d12-r11.25-randtok/evals/core.json deleted file mode 100644 index 48f06d21a6672fd3fe140f024b8b13e99fdeba94..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": 0.05120713426476866, - "core_results": { - "hellaswag_zeroshot": 0.2790280878543854, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.07145317643880844, - "arc_easy": 0.31607744097709656, - "arc_challenge": 0.20563139021396637, - "copa": 0.4899999797344208, - "commonsense_qa": 0.3054873049259186, - "piqa": 0.5429815053939819, - "openbook_qa": 0.23200000822544098, - "lambada_openai": 0.19153890013694763, - "hellaswag": 0.27853015065193176, - "winograd": 0.5604395866394043, - "winogrande": 0.49486976861953735, - "bigbench_dyck_languages": 0.08400000631809235, - "agi_eval_lsat_ar": 0.30869564414024353, - "bigbench_cs_algorithms": 0.4204545319080353, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.021948911249637604, - "coqa": 0.05273706465959549, - "boolq": 0.430886834859848, - "bigbench_language_identification": 0.2540999948978424 - }, - "centered_results": { - "hellaswag_zeroshot": 0.038704117139180504, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.07145317643880844, - "arc_easy": 0.08810325463612874, - "arc_challenge": -0.059158146381378174, - "copa": -0.020000040531158447, - "commonsense_qa": 0.1318591311573982, - "piqa": 0.08596301078796387, - "openbook_qa": -0.02399998903274536, - "lambada_openai": 0.19153890013694763, - "hellaswag": 0.03804020086924235, - "winograd": 0.1208791732788086, - "winogrande": -0.010260462760925293, - "bigbench_dyck_languages": 0.08400000631809235, - "agi_eval_lsat_ar": 0.13586955517530439, - "bigbench_cs_algorithms": 0.4204545319080353, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.021948911249637604, - "coqa": 0.05273706465959549, - "boolq": -0.4976662240530315, - "bigbench_language_identification": 0.17942793718134478 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-randtok/evals/samples.json b/experiments/clean1930s-d12-r11.25-randtok/evals/samples.json deleted file mode 100644 index 30566f62be7b55eea29b81cefb1504b54aa520c9..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the nation, and the capital of the nation is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as that of silver, and the latter is the same as the former" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and the day of the week will be Sunday. The day of the" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the most common form of the word. The word is used in the sense of" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the planets, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a very fine one, and I have never seen it in my life. I" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the parts of the body, and x is the number of the" - } - ], - "unconditioned_samples": [ - "<|bos|>ccliary on a certain phase of life-into which the crews of vessels of war have their chief and are fortunate in holding the chief place at the point of the stroke-what second seeks life artillery fortification in the future, the young shoot; and resisting a longer bullet-wounded tale in order to conform to the demands of business, declare openly, what is the duty of these young officers. 339\n\nINTRODUCTION\n\nAn indefatigable fellow who has attained middle life experiences abundant satisfaction no less keen on such occasions than when a fighter's son with ardour in the fray will, having crowned his freshness with laurels of strong individuality and shouting the", - "<|bos|>CHAPTER XVIII.\n\nPersonal Voyages in Shropshire.- Voyage to Lourdes.-Inland Voyages to Madison County. - Impressments upon the northern New\n\nHampshire Army.-The King entertained Chancellor Kent.-The French Mail-Scope.\n\n-Comparison of Narrative given by Prince Hoofloy. - Encounters and villages of the Church of Lourdes.-No Severance of Early Days.-The\n\nEnglish secure coal from Pilgrim Hall.--Mr. De Chaumier's\"-Public Windows.- Cholera of the Country.-(Robert Hall's)-Return to London before Alvare", - "<|bos|>PREFACE.\n\nix it in the seventh book to the lines full and true. ritie Mc-\n\nMullen, receiving theory kinde of his essay on \"The Heart and the Heart.\"\n\nSomebody has written of my scholarship, and mine ability. But no apology cately based on facts has beamed from the lips of Shobaugh: 'Tis ake win' and 'tis ake ake win\n\nBy early discovery eke of the common 'tall Hhor.'\n\nYoung Horace, who stands as fellow-musical with the immigration of\n\nClarkes and Iredellians; was born in the Pacific Alps", - "<|bos|>The company will draw on its rent at Five Pence.\n\nTail & Doors set up for Assoc. J. Glasgow Modern Greek\n\nSons, Parts 16 and 17.\n\nTECHNICAL PICTURES.\n\nIntroductory.\n\nThe writer of the foregoing points out two systems of pillaging as the result: one for intent and one for specialty, regardless of cost and charges; but as elements in analysis, the aims of the permisi Ph. and 3 years\n\nTail & Dooors loud. $ength. 61 a. 50 asses emmen. 25 phied ", - "<|bos|>ENDORION COURTISEMENT\n\nCOPYRIGHT, 1901, BY\n\nPHILIP S. ROGERS Esquire to the Honorable the Honorable the Board of Admiralty, Admiralty, &c.,:\n\n(MARY\n\nN. ROGERS Mary Heptam. The Hon. H. H.\n\nMackintosh.\n\nThe Honorable the Honorable the Honorable the Honorable the Queen Margot, and the Countess of\n\nPoultney, do certify that the following are the names of the persons named in the above-mentioned proposals, and the quantities of gold the said persons have Ibrahim the Harriethbout King etc., which", - "<|bos|>onica, 1. 3. 22 and weissen, 233, 4 Supr. 455.\n\nYAWARD, See BARPAIN ON.\n\n1 On my official report of my public duties in chief, Nays Lincoln, Boston, To-day, 31st December, 1861. [28th June, 1862] in\n\nWAR DEPARTMENT,\n\nWASHINGTON, June 10, 1863. \"MY DEAR BROTHER: By letter I received yesterday the letter written to you, father, father, and uncle, and sent you because you do not belong to them who wrote", - "<|bos|>AV-Government of the Six Nations, so long as it exists, is materially weak, and may be ultimately superseded, by the Government of Modern India itself; and a ministry in England may turn it in its administration into a formidable force, which it cannot cope with in India, as a subordinate agency, by extraordinary administrative talents, required for its focus; and in default may render futile any further offensive or nugatory effort to the natives of India. As a temporary agent, the Council put itself under the protection of\n\nJacob Grimm, an Edinburgh man, who, like him, was under orders.\n\nThe senior partner of the same", - "<|bos|>PREFACE.\n\nI scruple not to reiterate this impression, and to acquiesce, with grateful emotions, in the misfortunes which late, my dear Mr. Ballantyne, have Providence permitted for my own sake.\n\nIt is impossible that the author of a set of printed memoirs relating to Scotland could not find an excuse for having discovered his friend's address on a paper called undoubtedly \"The Popular Musick Services,\" published in the year-book of the year 1822 --a paper wholly in blank, for sale in Kensington Gardens, where it happened to have that particular merit, and on whose reception he thought it only right to use his name whenever" - ] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-randtok/evals/val_bpb.json b/experiments/clean1930s-d12-r11.25-randtok/evals/val_bpb.json deleted file mode 100644 index 66632cb7c068f4ceb610c63811c0921e572f2bf8..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.053600922794229 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25-randtok/run.json b/experiments/clean1930s-d12-r11.25-randtok/run.json deleted file mode 100644 index 7df4a3f4999cff938d74c156bedd87dddd4d16bd..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-randtok", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-randtok", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "175184ba2ee0f5d3", - "wandb_run_id": "6b8f19a5", - "created_at": 1784310529 -} diff --git a/experiments/clean1930s-d12-r11.25-randtok/summary.json b/experiments/clean1930s-d12-r11.25-randtok/summary.json deleted file mode 100644 index 0acee228c7f834705004a4bc0ea6cdf2c9a624bf..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-randtok", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25-randtok", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean-1930s", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.1128382305031206, - "minimum_sampled_val_bpb": 1.1128382305031206, - "full_val_bpb": 1.053600922794229, - "core_metric": 0.05120713426476866, - "centered_results": { - "hellaswag_zeroshot": 0.038704117139180504, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.07145317643880844, - "arc_easy": 0.08810325463612874, - "arc_challenge": -0.059158146381378174, - "copa": -0.020000040531158447, - "commonsense_qa": 0.1318591311573982, - "piqa": 0.08596301078796387, - "openbook_qa": -0.02399998903274536, - "lambada_openai": 0.19153890013694763, - "hellaswag": 0.03804020086924235, - "winograd": 0.1208791732788086, - "winogrande": -0.010260462760925293, - "bigbench_dyck_languages": 0.08400000631809235, - "agi_eval_lsat_ar": 0.13586955517530439, - "bigbench_cs_algorithms": 0.4204545319080353, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.021948911249637604, - "coqa": 0.05273706465959549, - "boolq": -0.4976662240530315, - "bigbench_language_identification": 0.17942793718134478 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the nation, and the capital of the nation is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as that of silver, and the latter is the same as the former" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and the day of the week will be Sunday. The day of the" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the most common form of the word. The word is used in the sense of" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the planets, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a very fine one, and I have never seen it in my life. I" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the parts of the body, and x is the number of the" - } - ], - "unconditioned_samples": [ - "<|bos|>ccliary on a certain phase of life-into which the crews of vessels of war have their chief and are fortunate in holding the chief place at the point of the stroke-what second seeks life artillery fortification in the future, the young shoot; and resisting a longer bullet-wounded tale in order to conform to the demands of business, declare openly, what is the duty of these young officers. 339\n\nINTRODUCTION\n\nAn indefatigable fellow who has attained middle life experiences abundant satisfaction no less keen on such occasions than when a fighter's son with ardour in the fray will, having crowned his freshness with laurels of strong individuality and shouting the", - "<|bos|>CHAPTER XVIII.\n\nPersonal Voyages in Shropshire.- Voyage to Lourdes.-Inland Voyages to Madison County. - Impressments upon the northern New\n\nHampshire Army.-The King entertained Chancellor Kent.-The French Mail-Scope.\n\n-Comparison of Narrative given by Prince Hoofloy. - Encounters and villages of the Church of Lourdes.-No Severance of Early Days.-The\n\nEnglish secure coal from Pilgrim Hall.--Mr. De Chaumier's\"-Public Windows.- Cholera of the Country.-(Robert Hall's)-Return to London before Alvare", - "<|bos|>PREFACE.\n\nix it in the seventh book to the lines full and true. ritie Mc-\n\nMullen, receiving theory kinde of his essay on \"The Heart and the Heart.\"\n\nSomebody has written of my scholarship, and mine ability. But no apology cately based on facts has beamed from the lips of Shobaugh: 'Tis ake win' and 'tis ake ake win\n\nBy early discovery eke of the common 'tall Hhor.'\n\nYoung Horace, who stands as fellow-musical with the immigration of\n\nClarkes and Iredellians; was born in the Pacific Alps", - "<|bos|>The company will draw on its rent at Five Pence.\n\nTail & Doors set up for Assoc. J. Glasgow Modern Greek\n\nSons, Parts 16 and 17.\n\nTECHNICAL PICTURES.\n\nIntroductory.\n\nThe writer of the foregoing points out two systems of pillaging as the result: one for intent and one for specialty, regardless of cost and charges; but as elements in analysis, the aims of the permisi Ph. and 3 years\n\nTail & Dooors loud. $ength. 61 a. 50 asses emmen. 25 phied ", - "<|bos|>ENDORION COURTISEMENT\n\nCOPYRIGHT, 1901, BY\n\nPHILIP S. ROGERS Esquire to the Honorable the Honorable the Board of Admiralty, Admiralty, &c.,:\n\n(MARY\n\nN. ROGERS Mary Heptam. The Hon. H. H.\n\nMackintosh.\n\nThe Honorable the Honorable the Honorable the Honorable the Queen Margot, and the Countess of\n\nPoultney, do certify that the following are the names of the persons named in the above-mentioned proposals, and the quantities of gold the said persons have Ibrahim the Harriethbout King etc., which", - "<|bos|>onica, 1. 3. 22 and weissen, 233, 4 Supr. 455.\n\nYAWARD, See BARPAIN ON.\n\n1 On my official report of my public duties in chief, Nays Lincoln, Boston, To-day, 31st December, 1861. [28th June, 1862] in\n\nWAR DEPARTMENT,\n\nWASHINGTON, June 10, 1863. \"MY DEAR BROTHER: By letter I received yesterday the letter written to you, father, father, and uncle, and sent you because you do not belong to them who wrote", - "<|bos|>AV-Government of the Six Nations, so long as it exists, is materially weak, and may be ultimately superseded, by the Government of Modern India itself; and a ministry in England may turn it in its administration into a formidable force, which it cannot cope with in India, as a subordinate agency, by extraordinary administrative talents, required for its focus; and in default may render futile any further offensive or nugatory effort to the natives of India. As a temporary agent, the Council put itself under the protection of\n\nJacob Grimm, an Edinburgh man, who, like him, was under orders.\n\nThe senior partner of the same", - "<|bos|>PREFACE.\n\nI scruple not to reiterate this impression, and to acquiesce, with grateful emotions, in the misfortunes which late, my dear Mr. Ballantyne, have Providence permitted for my own sake.\n\nIt is impossible that the author of a set of printed memoirs relating to Scotland could not find an excuse for having discovered his friend's address on a paper called undoubtedly \"The Popular Musick Services,\" published in the year-book of the year 1822 --a paper wholly in blank, for sale in Kensington Gardens, where it happened to have that particular merit, and on whose reception he thought it only right to use his name whenever" - ], - "training_time_seconds": 6277.177423000336, - "stage_training_flops": 1.0985538644638433e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.0985538644638433e+18, - "config_fingerprint": "175184ba2ee0f5d3", - "git_commit_sha": "118adeb2fc3350cf27ad635af372bd9b9410c0ef", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/6b8f19a5", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/clean1930s-d12-r11.25-randtok", - "dataset_fingerprint": "4af5759948d15ff0", - "tokenizer_fingerprint": "f01549dbe6e2aaa9", - "unique_train_tokens": 1275519304, - "effective_epochs": 0.970873786164196 -} diff --git a/experiments/clean1930s-d12-r11.25-randtok/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d12-r11.25-randtok/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 0de3544da519fa2a4ac3bd71b929f5ae5a3f417e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,20 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25-randtok", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768, - "sampling": "random", - "sampling_seed": 42 - }, - "created_at": 1784310546 -} diff --git a/experiments/clean1930s-d12-r11.25-randtok/tokenizer/token_bytes.pt b/experiments/clean1930s-d12-r11.25-randtok/tokenizer/token_bytes.pt deleted file mode 100644 index ea66127b8045ccb5e30dc43f57f9812d849460fb..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b9de5ef70a8805ae1e23ed607d05ab79cf92582f123294be962ca714ff0664ec -size 132649 diff --git a/experiments/clean1930s-d12-r11.25-randtok/tokenizer/tokenizer.pkl b/experiments/clean1930s-d12-r11.25-randtok/tokenizer/tokenizer.pkl deleted file mode 100644 index c0c31da99f03bdb9a149b516c2d48c10ec57a9bd..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25-randtok/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:21d846dc85121d501a10b42cacd77297a6358e43203f03c320f64c2c395ba894 -size 411592 diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_000500.json b/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_000500.json deleted file mode 100644 index e3f4ea070fbd4660e2685b3b42239d46bdb0a0dc..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,139 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25", - "val_bpb": 1.3406276768636045, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25", - "wandb_run_id": "764ab08e", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/config.json", - "tokenizer_fingerprint": "fa7b147cb0e4b5db", - "git_commit_sha": "083cd7f99484b5e894a23e4a093de6d339c412ea", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "00e7ae6b9109c167", - "artifact_path": "experiments/clean1930s-d12-r11.25" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "00e7ae6b9109c167" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3406276768636045, - "smooth_train_loss": 3.8475896185653875, - "total_training_time": 1308.1742749214172, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_001000.json b/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_001000.json deleted file mode 100644 index a49512a2120e8c2614992870f1c4f2afb7ff88c6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,139 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25", - "val_bpb": 1.2453749523249924, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25", - "wandb_run_id": "764ab08e", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/config.json", - "tokenizer_fingerprint": "fa7b147cb0e4b5db", - "git_commit_sha": "083cd7f99484b5e894a23e4a093de6d339c412ea", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "00e7ae6b9109c167", - "artifact_path": "experiments/clean1930s-d12-r11.25" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "00e7ae6b9109c167" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2453749523249924, - "smooth_train_loss": 3.5828240195598844, - "total_training_time": 2642.554317712784, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_001500.json b/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_001500.json deleted file mode 100644 index 96dd3dc4993bc92fb7d27680fae439ce2723eb0e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,139 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25", - "val_bpb": 1.1784689301618998, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25", - "wandb_run_id": "764ab08e", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/config.json", - "tokenizer_fingerprint": "fa7b147cb0e4b5db", - "git_commit_sha": "083cd7f99484b5e894a23e4a093de6d339c412ea", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "00e7ae6b9109c167", - "artifact_path": "experiments/clean1930s-d12-r11.25" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "00e7ae6b9109c167" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.1784689301618998, - "smooth_train_loss": 3.3444808841561646, - "total_training_time": 3977.2887001037598, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_002000.json b/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_002000.json deleted file mode 100644 index 4263a4e47ff3cf48cd0cbd573f064cb92c9d9362..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,139 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r11.25", - "val_bpb": 1.1385790516794956, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25", - "wandb_run_id": "764ab08e", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/config.json", - "tokenizer_fingerprint": "fa7b147cb0e4b5db", - "git_commit_sha": "083cd7f99484b5e894a23e4a093de6d339c412ea", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "00e7ae6b9109c167", - "artifact_path": "experiments/clean1930s-d12-r11.25" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "00e7ae6b9109c167" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.1385790516794956, - "smooth_train_loss": 3.19688287657793, - "total_training_time": 5312.674092292786, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_002362.json b/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_002362.json deleted file mode 100644 index 1c41191c4d687aa13902f6b143e76f406d04c2f1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,139 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "clean1930s-d12-r11.25", - "val_bpb": 1.1175529448105965, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r11.25", - "wandb_run_id": "764ab08e", - "wandb_group": "clean1930s-d12", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/base_checkpoints", - "experiment_id": "clean1930s-d12-r11.25", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r11.25/config.json", - "tokenizer_fingerprint": "fa7b147cb0e4b5db", - "git_commit_sha": "083cd7f99484b5e894a23e4a093de6d339c412ea", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r11.25", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "00e7ae6b9109c167", - "artifact_path": "experiments/clean1930s-d12-r11.25" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "00e7ae6b9109c167" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": 1.1175529448105965, - "smooth_train_loss": 3.1313785378510484, - "total_training_time": 6279.889491558075, - "stage_training_flops": 1098553864463843328, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1098553864463843328 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/model_000500.pt b/experiments/clean1930s-d12-r11.25/base_checkpoints/model_000500.pt deleted file mode 100644 index 0ceb2b4169ca2a7d5e8a0d9777522a5b96a0693d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4e818606e95665b2d93328ebac939eff1721dfddd5bc633c670d05644ec02b09 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/model_001000.pt b/experiments/clean1930s-d12-r11.25/base_checkpoints/model_001000.pt deleted file mode 100644 index 93685362662a8925999a9a1c22e31c3e7391e50a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6afac069374a34152cb01eff7142a0e285d70880bee793b10c0de073805a3864 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/model_001500.pt b/experiments/clean1930s-d12-r11.25/base_checkpoints/model_001500.pt deleted file mode 100644 index 012c06c2b1386831b274f6e87ddf44c548057d06..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1da723ae288e3aedd7142082491758bd1a6be3791cb4b881f32909c63e22ec68 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/model_002000.pt b/experiments/clean1930s-d12-r11.25/base_checkpoints/model_002000.pt deleted file mode 100644 index 0b479dd3d324ef8af976ec600f87afe56ada089e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:108865e4d197f8c5f893fc9e0ef7568bf33258fa908fd843002e4033436e6169 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/model_002362.pt b/experiments/clean1930s-d12-r11.25/base_checkpoints/model_002362.pt deleted file mode 100644 index e349db1b9a56456c799c2d3ec26bc6805cd6edff..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:66478807ab543dc52a021b0989d39b1f59fdb9978bc679dc23781c97b0136a23 -size 792761690 diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_000500_rank0.pt b/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 2195a9fc6b51e60c259798dd16e2305120faf409..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a57bc43f0fc66aa44a67432373a76db9ee4be1b9b425ec037327188641df180d -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_001000_rank0.pt b/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index ddf177d9fce73af3f8558903ccd6c4aad0fe481b..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5876a82a2083edec62f32471e1e8cd87749d59d9663905c7762447601e480bb7 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_001500_rank0.pt b/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index c8169d0f893f94927b95381eea9124869b18e62f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:46edf74481ffc6508a1abdf1ad0837b2b1d1fefb6d09328fed8067b7819df19a -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_002000_rank0.pt b/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 42c474cb7a1c58bb6019fa157a31e55eff6188a6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c13b9d6a8e454e20f62c5f1ea69c9076a2baffe739b3af61569b029c3203fe33 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_002362_rank0.pt b/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 1d53108e4c979b5a82cb8068289d4e4b66c3c274..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a09823e14e88be9d6acaea7b9d3a2e5c88b435d1d40bbcbe656f780c874cdb32 -size 1246165357 diff --git a/experiments/clean1930s-d12-r11.25/config.json b/experiments/clean1930s-d12-r11.25/config.json deleted file mode 100644 index 579f1648757616da1f6cde7bce72504ad8d6b386..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/config.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r11.25", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r11.25", - "group": "clean1930s-d12", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "00e7ae6b9109c167", - "artifact_path": "experiments/clean1930s-d12-r11.25" -} diff --git a/experiments/clean1930s-d12-r11.25/evals/core.json b/experiments/clean1930s-d12-r11.25/evals/core.json deleted file mode 100644 index 264ffc5cda224fa6fdf951ac0fd176d7cdf8429d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": 0.06238580602029931, - "core_results": { - "hellaswag_zeroshot": 0.2756423056125641, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.07450420409440994, - "arc_easy": 0.32281145453453064, - "arc_challenge": 0.21587030589580536, - "copa": 0.5399999618530273, - "commonsense_qa": 0.26535627245903015, - "piqa": 0.5402611494064331, - "openbook_qa": 0.242000013589859, - "lambada_openai": 0.2115272581577301, - "hellaswag": 0.2757418751716614, - "winograd": 0.5641025900840759, - "winogrande": 0.49723756313323975, - "bigbench_dyck_languages": 0.1120000034570694, - "agi_eval_lsat_ar": 0.2913043200969696, - "bigbench_cs_algorithms": 0.41969695687294006, - "bigbench_operators": 0.06190476566553116, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.014758750796318054, - "coqa": 0.05574345588684082, - "boolq": 0.4880733788013458, - "bigbench_language_identification": 0.2556000053882599 - }, - "centered_results": { - "hellaswag_zeroshot": 0.034189740816752114, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.07450420409440994, - "arc_easy": 0.09708193937937419, - "arc_challenge": -0.04550625880559286, - "copa": 0.07999992370605469, - "commonsense_qa": 0.08169534057378768, - "piqa": 0.08052229881286621, - "openbook_qa": -0.010666648546854654, - "lambada_openai": 0.2115272581577301, - "hellaswag": 0.034322500228881836, - "winograd": 0.12820518016815186, - "winogrande": -0.005524873733520508, - "bigbench_dyck_languages": 0.1120000034570694, - "agi_eval_lsat_ar": 0.11413040012121199, - "bigbench_cs_algorithms": 0.41969695687294006, - "bigbench_operators": 0.06190476566553116, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.014758750796318054, - "coqa": 0.05574345588684082, - "boolq": -0.34717531894382675, - "bigbench_language_identification": 0.18107811373845972 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25/evals/samples.json b/experiments/clean1930s-d12-r11.25/evals/samples.json deleted file mode 100644 index c7289a8e77a728a06536f0b3a4b89a7913385f1a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the Empire, and the capital of the Empire is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the \"milk of the earth,\" and the \"milk of" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If yesterday was Saturday, then tomorrow will be Saturday. If yesterday was" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the hot, and the hot is the cold. The hot is the hot," - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: 1. The sun, 2. The moon, 3. The" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the red of the red of the red of the red of the red of the" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of days in which the earth is in motion, and the number of" - } - ], - "unconditioned_samples": [ - "<|bos|>CHAPTER V FORMS ON STEAM MILLS, It has already been pointed out that the very good Anglo- Saxon capital and its effects are of recent introduction in the East. The consciousness of modernity in architecture is rapidly \"tasking.\" In Europe and America, no single innovation, no single change in the least tenuous architecture, would be tolerated. On the contrary, so many hardy pioneers of civilization now lie dead in the mud or are idly rotting in stones; so general an application would still be to a period prior to 1870, just as we now to a century since, are inclined to reckon an ignorant man", - "<|bos|>PREFACE.\n\nso often quoted and described, and revealed the fact of average, almost unknowable, within the little knowable field of opinion of his.\n\nLet me hope that we may go as far as the narrative will admit into the explanation of such expressions as The Genealogies of Defendant, The Book-hunter, The Wise Man, etc.\n\nIt is often said, and so I deem it, that the gentleman who draws the following is not moving on his account, and does not design the question.\n\nLet us say, to the reader, This pantheistic text furnishes a full and complete explanation of the about the differences which make", - "<|bos|>PREFACE.\n\nix emphatic word is my wish. with one or two reasonable exceptions, and who experiences any emotion or fear on hearing of a narrative of her \"Franks.\"\n\nA strong and authoritative book, on a vouchsafed subject, or novelty or tedium to human beliefes careles of this life of yours shall be the common wish of every man, and it will make the incipient man that child alone that compilers, with the fore-power of vision, have by indolence, or vice, or other powerful means, precipitated the history ofori- this society.\"MI- MENTS FROM THE JUDGmeign. I", - "<|bos|> Dreams and Reverses, by Rev. P. I. Sheard.\n\n(ACT IV.) WALL RUN RECREATIONS\n\nIII Last Summer Evening, I took a Narrow Way to Princeton, Indiana, the night after Shamokoka fired their last shot, in their settlement and near Spring Hill, near\n\nAkosta Flats. Throughout the day whitening the trees around, tinged the spires of farms seated on clumps of trees and in all directions about them, and we had a fine time. Evidently to begin with, there came a large party of our men out of the lines, among", - "<|bos|>Dieta, i. 653, 686, 692.\n\nCook's ear, i. 76-80, 167, 235, 236.\n\nDoor-watchage, i. 82, 83, 113.\n\nDoolittle's reading, i. 548.\n\nDoomer's simply court notice, i. 355.\n\nEhambra parting teeth, i. 700, 702.. Ehambra weeping, i. 146. e.\n\nEast Bangalore, ii. 374.\n\nEcclesiastics", - "<|bos|>orms, while the moral principle is allowable in ordinary life. The boys belonging to a tradesman's home are let alone after spending a few weeks, during which he does not receive any pay; this exhausts their labors. Since there are no other means for carrying through business, few will be willing to undertake it; but few display the ability that can sustain sustained excitement by the prompt bestowal of a generous payment.\n\nThe third day is occupied in operaucb Reading. Sticks have been hired for over six months for delivery at a drug store, but this payment has been withheld because otherwise (Wat v. Ortsville", - "<|bos|>) Ruth and Robin lick'll look an' put up o' Tal-ledge's our followlin' lar, Iween, a 're luther Tol a little afraid 'll stay 'ead a-dandy at a P'ock harness an' es 'to Awlfe Cl Proper should slavery me sent, Weritan To! Cane flax can Gither it ble at Cutnish Maalock the times of aeething, Some lady misses to be able to Pane the vaeuce ground, some lady rod theows hast' a to Pate downt wi the", - "<|bos|> shares ought to be limited by the terms of a deed and however they may be in the hands of numerous individuals, the sum can easily be made large by even a few executed persons alone. Let the number of such persons gathered and the results produced justify the conclusion that the shares were free from all liabilities in their hands, and that their share had a legal amount sufficient to cover them at all times.\n\nThe rule adopted by the American Accountant General for Improving the Powers of the President of the United States Government in regard to the use of legal instruments for the transaction of business in the execution of contracts, applies as well to such legal" - ] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25/evals/val_bpb.json b/experiments/clean1930s-d12-r11.25/evals/val_bpb.json deleted file mode 100644 index 36642bfe6a681b55ed411f548a6c880b2e611c22..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.0596072239127994 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json b/experiments/clean1930s-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json deleted file mode 100644 index 4111f11a1e12fe1407f26ba775d0fab4d9087f93..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.06282022394172 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25/evals/val_bpb_on_think-dataset.json b/experiments/clean1930s-d12-r11.25/evals/val_bpb_on_think-dataset.json deleted file mode 100644 index ba1f719b5724a5f39190b16170e12252c6519481..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/evals/val_bpb_on_think-dataset.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.1058237779626336 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r11.25/run.json b/experiments/clean1930s-d12-r11.25/run.json deleted file mode 100644 index 5b6360678536c9e2ae96a6b1160a4595bcf3534a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "00e7ae6b9109c167", - "wandb_run_id": "764ab08e", - "created_at": 1783954179 -} diff --git a/experiments/clean1930s-d12-r11.25/summary.json b/experiments/clean1930s-d12-r11.25/summary.json deleted file mode 100644 index bd1692d698923853bfe8658980af0643d1cb650a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r11.25", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean-1930s", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.1175529448105965, - "minimum_sampled_val_bpb": 1.1175529448105965, - "full_val_bpb": 1.0596072239127994, - "core_metric": 0.06238580602029931, - "centered_results": { - "hellaswag_zeroshot": 0.034189740816752114, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.07450420409440994, - "arc_easy": 0.09708193937937419, - "arc_challenge": -0.04550625880559286, - "copa": 0.07999992370605469, - "commonsense_qa": 0.08169534057378768, - "piqa": 0.08052229881286621, - "openbook_qa": -0.010666648546854654, - "lambada_openai": 0.2115272581577301, - "hellaswag": 0.034322500228881836, - "winograd": 0.12820518016815186, - "winogrande": -0.005524873733520508, - "bigbench_dyck_languages": 0.1120000034570694, - "agi_eval_lsat_ar": 0.11413040012121199, - "bigbench_cs_algorithms": 0.41969695687294006, - "bigbench_operators": 0.06190476566553116, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.014758750796318054, - "coqa": 0.05574345588684082, - "boolq": -0.34717531894382675, - "bigbench_language_identification": 0.18107811373845972 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the Empire, and the capital of the Empire is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the \"milk of the earth,\" and the \"milk of" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If yesterday was Saturday, then tomorrow will be Saturday. If yesterday was" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the hot, and the hot is the cold. The hot is the hot," - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: 1. The sun, 2. The moon, 3. The" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the red of the red of the red of the red of the red of the" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of days in which the earth is in motion, and the number of" - } - ], - "unconditioned_samples": [ - "<|bos|>CHAPTER V FORMS ON STEAM MILLS, It has already been pointed out that the very good Anglo- Saxon capital and its effects are of recent introduction in the East. The consciousness of modernity in architecture is rapidly \"tasking.\" In Europe and America, no single innovation, no single change in the least tenuous architecture, would be tolerated. On the contrary, so many hardy pioneers of civilization now lie dead in the mud or are idly rotting in stones; so general an application would still be to a period prior to 1870, just as we now to a century since, are inclined to reckon an ignorant man", - "<|bos|>PREFACE.\n\nso often quoted and described, and revealed the fact of average, almost unknowable, within the little knowable field of opinion of his.\n\nLet me hope that we may go as far as the narrative will admit into the explanation of such expressions as The Genealogies of Defendant, The Book-hunter, The Wise Man, etc.\n\nIt is often said, and so I deem it, that the gentleman who draws the following is not moving on his account, and does not design the question.\n\nLet us say, to the reader, This pantheistic text furnishes a full and complete explanation of the about the differences which make", - "<|bos|>PREFACE.\n\nix emphatic word is my wish. with one or two reasonable exceptions, and who experiences any emotion or fear on hearing of a narrative of her \"Franks.\"\n\nA strong and authoritative book, on a vouchsafed subject, or novelty or tedium to human beliefes careles of this life of yours shall be the common wish of every man, and it will make the incipient man that child alone that compilers, with the fore-power of vision, have by indolence, or vice, or other powerful means, precipitated the history ofori- this society.\"MI- MENTS FROM THE JUDGmeign. I", - "<|bos|> Dreams and Reverses, by Rev. P. I. Sheard.\n\n(ACT IV.) WALL RUN RECREATIONS\n\nIII Last Summer Evening, I took a Narrow Way to Princeton, Indiana, the night after Shamokoka fired their last shot, in their settlement and near Spring Hill, near\n\nAkosta Flats. Throughout the day whitening the trees around, tinged the spires of farms seated on clumps of trees and in all directions about them, and we had a fine time. Evidently to begin with, there came a large party of our men out of the lines, among", - "<|bos|>Dieta, i. 653, 686, 692.\n\nCook's ear, i. 76-80, 167, 235, 236.\n\nDoor-watchage, i. 82, 83, 113.\n\nDoolittle's reading, i. 548.\n\nDoomer's simply court notice, i. 355.\n\nEhambra parting teeth, i. 700, 702.. Ehambra weeping, i. 146. e.\n\nEast Bangalore, ii. 374.\n\nEcclesiastics", - "<|bos|>orms, while the moral principle is allowable in ordinary life. The boys belonging to a tradesman's home are let alone after spending a few weeks, during which he does not receive any pay; this exhausts their labors. Since there are no other means for carrying through business, few will be willing to undertake it; but few display the ability that can sustain sustained excitement by the prompt bestowal of a generous payment.\n\nThe third day is occupied in operaucb Reading. Sticks have been hired for over six months for delivery at a drug store, but this payment has been withheld because otherwise (Wat v. Ortsville", - "<|bos|>) Ruth and Robin lick'll look an' put up o' Tal-ledge's our followlin' lar, Iween, a 're luther Tol a little afraid 'll stay 'ead a-dandy at a P'ock harness an' es 'to Awlfe Cl Proper should slavery me sent, Weritan To! Cane flax can Gither it ble at Cutnish Maalock the times of aeething, Some lady misses to be able to Pane the vaeuce ground, some lady rod theows hast' a to Pate downt wi the", - "<|bos|> shares ought to be limited by the terms of a deed and however they may be in the hands of numerous individuals, the sum can easily be made large by even a few executed persons alone. Let the number of such persons gathered and the results produced justify the conclusion that the shares were free from all liabilities in their hands, and that their share had a legal amount sufficient to cover them at all times.\n\nThe rule adopted by the American Accountant General for Improving the Powers of the President of the United States Government in regard to the use of legal instruments for the transaction of business in the execution of contracts, applies as well to such legal" - ], - "training_time_seconds": 6279.889491558075, - "stage_training_flops": 1.0985538644638433e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.0985538644638433e+18, - "config_fingerprint": "00e7ae6b9109c167", - "git_commit_sha": "083cd7f99484b5e894a23e4a093de6d339c412ea", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/764ab08e", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/clean1930s-d12-r11.25", - "dataset_fingerprint": "4af5759948d15ff0", - "tokenizer_fingerprint": "fa7b147cb0e4b5db", - "unique_train_tokens": 1275519304, - "effective_epochs": 0.970873786164196 -} diff --git a/experiments/clean1930s-d12-r11.25/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d12-r11.25/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 1f01bf637a5554656b2e99480af66a9565d17ef7..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r11.25", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1783954241 -} diff --git a/experiments/clean1930s-d12-r11.25/tokenizer/token_bytes.pt b/experiments/clean1930s-d12-r11.25/tokenizer/token_bytes.pt deleted file mode 100644 index 26d7e6581c41a049b038de82b8bb65a1c8d57c8f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4d25a126690db557828b34dc3bd3e8ebd615058016c1870e76b32b31266805ed -size 132649 diff --git a/experiments/clean1930s-d12-r11.25/tokenizer/tokenizer.pkl b/experiments/clean1930s-d12-r11.25/tokenizer/tokenizer.pkl deleted file mode 100644 index d21e36e06c64fe9e86d8d579bccaae5542ad5ac9..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r11.25/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4932125f789267dbd571cc4bd19339ecfd7160c9b9cc0441ba6635756e4a8b0e -size 407747 diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_000500.json b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_000500.json deleted file mode 100644 index 7fdbf177e459ceb0f896d2f649bc42a2ce23b414..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,222 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r12-ctx4096", - "val_bpb": 1.3394909610264625, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r12-ctx4096", - "wandb_run_id": "e564d552", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio12,ctx4096,midtrain-schedule,fp8", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "23031c7d2f7cc08da131639554d952aef02069e9", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r12-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r12-ctx4096", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8" - ] - }, - "config_fingerprint": "f54a41c1b7b43ef4", - "artifact_path": "experiments/clean1930s-d12-r12-ctx4096" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r12-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "f54a41c1b7b43ef4", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769, - "mixture": { - "cursors": { - "original": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 262176768, - "source_tokens": { - "original": 262176768, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 1.3394909610264625, - "smooth_train_loss": 3.7008363848340475, - "total_training_time": 1494.5915577411652, - "stage_training_flops": 291921016651776000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 291921016651776000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_001000.json b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_001000.json deleted file mode 100644 index eb85a501067ae9a9bbfaef5feb445ac054b9c377..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,222 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r12-ctx4096", - "val_bpb": 1.2353599635607029, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r12-ctx4096", - "wandb_run_id": "e564d552", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio12,ctx4096,midtrain-schedule,fp8", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "23031c7d2f7cc08da131639554d952aef02069e9", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r12-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r12-ctx4096", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8" - ] - }, - "config_fingerprint": "f54a41c1b7b43ef4", - "artifact_path": "experiments/clean1930s-d12-r12-ctx4096" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r12-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "f54a41c1b7b43ef4", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769, - "mixture": { - "cursors": { - "original": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 524320768, - "source_tokens": { - "original": 524320768, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 1.2353599635607029, - "smooth_train_loss": 3.5236556054731762, - "total_training_time": 3021.539851665497, - "stage_training_flops": 583842033303552000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 583842033303552000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_001500.json b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_001500.json deleted file mode 100644 index fb77d29275aeb07f42cbaeaa74478e07b4716356..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,222 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r12-ctx4096", - "val_bpb": 1.1534062642313452, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r12-ctx4096", - "wandb_run_id": "e564d552", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio12,ctx4096,midtrain-schedule,fp8", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 1000, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "23031c7d2f7cc08da131639554d952aef02069e9", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r12-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r12-ctx4096", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8" - ] - }, - "config_fingerprint": "f54a41c1b7b43ef4", - "artifact_path": "experiments/clean1930s-d12-r12-ctx4096" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r12-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "f54a41c1b7b43ef4", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86521538, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86521538, - "mixture": { - "cursors": { - "original": { - "file_idx": 7, - "pos": 86521538, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86521538 - }, - "midtrain_r30": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 786497536, - "source_tokens": { - "original": 786497536, - "midtrain_r30": 0, - "midtrain_r60": 0 - }, - "active_stage_idx": 0 - } - }, - "loop_state": { - "min_val_bpb": 1.1534062642313452, - "smooth_train_loss": 3.3564800354053057, - "total_training_time": 4609.387755393982, - "stage_training_flops": 875763049955328000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 875763049955328000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_002000.json b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_002000.json deleted file mode 100644 index 0bd69ef52ec604d9dac8540f7af1fda94f1d8cf0..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,222 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "clean1930s-d12-r12-ctx4096", - "val_bpb": 1.1278678135192355, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r12-ctx4096", - "wandb_run_id": "e564d552", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio12,ctx4096,midtrain-schedule,fp8", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 1000, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "23031c7d2f7cc08da131639554d952aef02069e9", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r12-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r12-ctx4096", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8" - ] - }, - "config_fingerprint": "f54a41c1b7b43ef4", - "artifact_path": "experiments/clean1930s-d12-r12-ctx4096" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r12-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "f54a41c1b7b43ef4", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 1, - "pos": 23801282, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 23801282, - "mixture": { - "cursors": { - "original": { - "file_idx": 9, - "pos": 24872256, - "epoch": 1, - "pq_idx": 9, - "rg_idx": 24872256 - }, - "midtrain_r30": { - "file_idx": 1, - "pos": 23801282, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 23801282 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 1048641536, - "source_tokens": { - "original": 924844032, - "midtrain_r30": 123797504, - "midtrain_r60": 0 - }, - "active_stage_idx": 1 - } - }, - "loop_state": { - "min_val_bpb": 1.1278678135192355, - "smooth_train_loss": 3.2400561803845425, - "total_training_time": 6135.895956039429, - "stage_training_flops": 1167684066607104000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1167684066607104000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_002500.json b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_002500.json deleted file mode 100644 index a1780493a2f25f22224217df8825171816c48374..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_002500.json +++ /dev/null @@ -1,222 +0,0 @@ -{ - "step": 2500, - "training_complete": false, - "experiment_id": "clean1930s-d12-r12-ctx4096", - "val_bpb": 1.0910322707751114, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r12-ctx4096", - "wandb_run_id": "e564d552", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio12,ctx4096,midtrain-schedule,fp8", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 1000, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "23031c7d2f7cc08da131639554d952aef02069e9", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r12-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r12-ctx4096", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8" - ] - }, - "config_fingerprint": "f54a41c1b7b43ef4", - "artifact_path": "experiments/clean1930s-d12-r12-ctx4096" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r12-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "f54a41c1b7b43ef4", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 1, - "pos": 21704066, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 21704066, - "mixture": { - "cursors": { - "original": { - "file_idx": 9, - "pos": 24872256, - "epoch": 1, - "pq_idx": 9, - "rg_idx": 24872256 - }, - "midtrain_r30": { - "file_idx": 2, - "pos": 64249216, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 64249216 - }, - "midtrain_r60": { - "file_idx": 1, - "pos": 21704066, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 21704066 - } - }, - "cumulative_tokens": 1310785536, - "source_tokens": { - "original": 924844032, - "midtrain_r30": 264241152, - "midtrain_r60": 121700352 - }, - "active_stage_idx": 2 - } - }, - "loop_state": { - "min_val_bpb": 1.0910322707751114, - "smooth_train_loss": 3.1627455750850237, - "total_training_time": 7662.492534160614, - "stage_training_flops": 1459605083258880000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1459605083258880000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_002520.json b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_002520.json deleted file mode 100644 index 4398eed8665ad952f4718e0d6d8070b7e5b73feb..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/meta_002520.json +++ /dev/null @@ -1,222 +0,0 @@ -{ - "step": 2520, - "training_complete": true, - "experiment_id": "clean1930s-d12-r12-ctx4096", - "val_bpb": 1.0901240605951916, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d12-r12-ctx4096", - "wandb_run_id": "e564d552", - "wandb_group": "clean1930s-d12-ctx4096", - "wandb_tags": "think-dataset-clean-1930s,d12,ratio12,ctx4096,midtrain-schedule,fp8", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 1000, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok", - "mixture_source_dirs": "{\"original\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_original\", \"midtrain_r30\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_midtrain_r30\", \"midtrain_r60\": \"/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/pretok_midtrain_r60\"}", - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d12-r12-ctx4096/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "23031c7d2f7cc08da131639554d952aef02069e9", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d12-r12-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r12-ctx4096", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8" - ] - }, - "config_fingerprint": "f54a41c1b7b43ef4", - "artifact_path": "experiments/clean1930s-d12-r12-ctx4096" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d12-r12-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "f54a41c1b7b43ef4", - "fp8_unavailable_reason": "NVIDIA A100-SXM4-40GB is SM 80; FP8 requires SM 89+ (Ada/Hopper)" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 1, - "pos": 32190146, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 32190146, - "mixture": { - "cursors": { - "original": { - "file_idx": 9, - "pos": 24872256, - "epoch": 1, - "pq_idx": 9, - "rg_idx": 24872256 - }, - "midtrain_r30": { - "file_idx": 2, - "pos": 64249216, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 64249216 - }, - "midtrain_r60": { - "file_idx": 1, - "pos": 32190146, - "epoch": 1, - "pq_idx": 1, - "rg_idx": 32190146 - } - }, - "cumulative_tokens": 1321271296, - "source_tokens": { - "original": 924844032, - "midtrain_r30": 264241152, - "midtrain_r60": 132186112 - }, - "active_stage_idx": 2 - } - }, - "loop_state": { - "min_val_bpb": 1.0901240605951916, - "smooth_train_loss": 3.1538503603170533, - "total_training_time": 7723.430352926254, - "stage_training_flops": 1471281923924951040, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1471281923924951040 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_000500.pt b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_000500.pt deleted file mode 100644 index 9cf7ecd87b045a3000232070ebc4590135d45489..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6a02d598c92e1ad5b765d3d0eb420db85489ea39645054c5fa07b5c1064a0d98 -size 792761690 diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_001000.pt b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_001000.pt deleted file mode 100644 index c4436a9d653f8c39698de78339eec870008e802f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:76925cc051b220fef8022928c1be79ff33cc8c645bac3d9c99330a5cd0481b4f -size 792761690 diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_001500.pt b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_001500.pt deleted file mode 100644 index 7feb80a0904e958ca32da58c0e398f8c744a5849..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:ae4ba119454fbb2e5a9002c6b085cf6defd58e9e56d0020b1afe48e651ceced8 -size 792761690 diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_002000.pt b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_002000.pt deleted file mode 100644 index 313f8416c37c4600b040ce20b0111ffca2e5f0ab..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a8f34df504598c38686c112f21609ef2c4782e1a2a9d49e752f5eee461f873ba -size 792761690 diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_002500.pt b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_002500.pt deleted file mode 100644 index 547436e6f79d03a4bd7ba2927d0fc00b37538efd..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:22ec550702fc1f867fe73c310471c6a1c1bcfaa11dd2384c9276d6c84f389c90 -size 792761690 diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_002520.pt b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_002520.pt deleted file mode 100644 index dfb4d65d9679408f33e4a4cf39bb64e165de475c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/model_002520.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b15ba4053c9c3165988bd8bb70adae3937402f8d8b63c3c592d8b1619de2ee9b -size 792761690 diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_000500_rank0.pt b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 86a28e94665555ad5b6b5cdab78b7dfec2ae711a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d99ad5c617d963fd673cd81134e61bdc8163dbd86a268357ad18f6eb68803dfd -size 1246165357 diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_001000_rank0.pt b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 66c3a6b28275cd7da3f10db8a9eb8283edeb536a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9c3efcacd3be64a437fb00c2d8d58ffece9581d9e3f8f53bcfba41d762ead912 -size 1246165357 diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_001500_rank0.pt b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index ff49188e5e47d1b50f5d29e71bc4ca49a9b931db..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:969e98e437d5e82ec160b8bac637a4f50efc79f622ef4c86328c70a38a7471fd -size 1246165357 diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_002000_rank0.pt b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 9459da8971afae1f34f00ce4465c995d40e5e904..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7139630f8def9773169031966aa4c00e2daf5f66d30a83685e92368f0a6e60ff -size 1246165357 diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_002500_rank0.pt b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_002500_rank0.pt deleted file mode 100644 index f65013e9aae7e61e655ffc355e1f3c7f06432a3f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f40de772fe8dab02d5de78f4c260b56321424a91c3d24bfc5fb095590bff9d61 -size 1246165357 diff --git a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_002520_rank0.pt b/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_002520_rank0.pt deleted file mode 100644 index c4555c974380a6c1d1a33210a0105433d2803831..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/base_checkpoints/optim_002520_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:69b60e67ecc3b8ab39ccb6f76883c53e6e1546c4c3484a77f32e8fa33095eb36 -size 1246165357 diff --git a/experiments/clean1930s-d12-r12-ctx4096/config.json b/experiments/clean1930s-d12-r12-ctx4096/config.json deleted file mode 100644 index d0b1b7a1995474ac810eb41efe80d1d1c137f03c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/config.json +++ /dev/null @@ -1,105 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d12-r12-ctx4096", - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 78, - "num_train_shards": 40, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 101, - "num_train_shards": 20, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 1321205760, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 924844032, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 1189085184, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d12-r12-ctx4096", - "group": "clean1930s-d12-ctx4096", - "tags": [ - "think-dataset-clean-1930s", - "d12", - "ratio12", - "ctx4096", - "midtrain-schedule", - "fp8" - ] - }, - "config_fingerprint": "f54a41c1b7b43ef4", - "artifact_path": "experiments/clean1930s-d12-r12-ctx4096" -} diff --git a/experiments/clean1930s-d12-r12-ctx4096/run.json b/experiments/clean1930s-d12-r12-ctx4096/run.json deleted file mode 100644 index 423e261c0c21af853f3b23fd29c3a99f215a680f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "clean1930s-d12-r12-ctx4096", - "stage": "base", - "base_experiment_id": "clean1930s-d12-r12-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "f54a41c1b7b43ef4", - "wandb_run_id": "e564d552", - "created_at": 1784920582 -} diff --git a/experiments/clean1930s-d12-r12-ctx4096/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d12-r12-ctx4096/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/clean1930s-d12-r12-ctx4096/tokenizer/token_bytes.pt b/experiments/clean1930s-d12-r12-ctx4096/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/clean1930s-d12-r12-ctx4096/tokenizer/tokenizer.pkl b/experiments/clean1930s-d12-r12-ctx4096/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d12-r12-ctx4096/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/meta_000500.json b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/meta_000500.json deleted file mode 100644 index 01421c0127a8faa5264a9a81f35d49f347ecb8bf..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,145 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "val_bpb": 1.1871477110433892, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "L" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "wandb_run_id": "14f8d6ff", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "97f9a8373c6852c594912151249caf4b8a9ee998", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "fp8" - ] - }, - "config_fingerprint": "67b81539f6613d04", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "67b81539f6613d04" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24322769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24322769 - }, - "loop_state": { - "min_val_bpb": 1.1871477110433892, - "smooth_train_loss": 3.3945704550936946, - "total_training_time": 999.2132678031921, - "stage_training_flops": 3245763761012736000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 3245763761012736000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/model_000500.pt b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/model_000500.pt deleted file mode 100644 index 705d7e6c238a2cef6a6e113a35be7f1f4e5dfdc6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:54be6eedde558d22722ce09563434b9890e9702841b45b1b698f8ac33daa7025 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index d6019132406a0f07ed7e34992c9511d9ea162817..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c7766244d86d99a2cad1b9f196413a0e193213635e2f6bc85ae91a72fec023f4 -size 717465265 diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank1.pt deleted file mode 100644 index 9140b8401d949264927d1f48a260ba7804dfebc4..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:85ad482372c1535df095998bf353351c17eaa2e5a7002141025c9405167de291 -size 717465265 diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank2.pt deleted file mode 100644 index 4117c07abcb1237709d458625c0b82bb92727fba..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc264d206ff72630f771afe9f7c8faff53f5367b27b2134baa05b9390daf3d03 -size 717465265 diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank3.pt deleted file mode 100644 index 9992d0354940b9ac6d9ad26b0aa36fe0d952c12e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:ac1bfb557caf498b940743b64a242b891c7d9744301c14214369c15990734781 -size 717465265 diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank4.pt b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank4.pt deleted file mode 100644 index bc6a24351caf8478c93d63a56c0dc77b782a25a9..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank4.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1020ca5e6e2c633a755a5c49e9d38c5bf560ba4482f6db2634ac11f78daf8793 -size 717465265 diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank5.pt b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank5.pt deleted file mode 100644 index 8d76b86d2353d861239b5c77c292f5d6f63a2faf..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank5.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2bbcbfa19ab1f94f0b4e5bba16ec95e31c342b55744701f1f9e2d05baba443a7 -size 717465265 diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank6.pt b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank6.pt deleted file mode 100644 index cc83cd5d82ff73432754e7dafc63e09957b901ae..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank6.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f80a5b1ebe129b2e08eaaa61afb84960acad1b07e1d19987f488fec6c43d7350 -size 717465265 diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank7.pt b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank7.pt deleted file mode 100644 index 44bff9e9d2a7fea38b927bcffc1fc6f122d3d1e5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/base_checkpoints/optim_000500_rank7.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d7d2640548eb206eeaf08804b69281500de1307810847c75cdfb3bb62a694872 -size 717465265 diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/config.json b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/config.json deleted file mode 100644 index b7ac24fafc268f4fcf39ab10ee855c45d01763f6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/config.json +++ /dev/null @@ -1,62 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "fp8" - ] - }, - "config_fingerprint": "67b81539f6613d04", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-fulltok-v1" -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/run.json b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/run.json deleted file mode 100644 index a47fef12476f81783db90438111e383b73812ca5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "67b81539f6613d04", - "wandb_run_id": "14f8d6ff", - "created_at": 1784129314 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/tokenizer/token_bytes.pt b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/tokenizer/tokenizer.pkl b/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-fulltok-v1/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_006500.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_006500.json deleted file mode 100644 index 1a9c4f263598619e5dcb93f6cdbc43e7de15b603..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_006500.json +++ /dev/null @@ -1,231 +0,0 @@ -{ - "step": 6500, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 6000, - "val_bpb": 0.9247502060081154, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "wandb_run_id": "d0cda484", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 8352, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "init_from_step": 6000, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok", - "mixture_source_dirs": "{\"midtrain_r30\": \"/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "640d2e56d9ca333095a9e23dcf54d498341552e8", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "branch": { - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_step": 6000, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 140, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 170, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 90, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 8757706752, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 6129975296, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 7881097216, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "43f2fdfbde0b68d4", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 6000, - "config_fingerprint": "43f2fdfbde0b68d4" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24423073, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24423073, - "mixture": { - "cursors": { - "midtrain_r30": { - "file_idx": 5, - "pos": 24423073, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24423073 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 6815875072, - "source_tokens": { - "original": 0, - "midtrain_r30": 524419072, - "midtrain_r60": 0 - }, - "active_stage_idx": 1 - } - }, - "loop_state": { - "min_val_bpb": 0.9247502060081154, - "smooth_train_loss": 2.6664623196737285, - "total_training_time": 1108.3150436878204, - "stage_start_step": 6000, - "stage_training_flops": 2711401109913600000, - "inherited_parent_flops": 3.25368133189632e+19, - "cumulative_pipeline_training_flops": 3.52482144288768e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_007000.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_007000.json deleted file mode 100644 index e63c027736cd78a791fe6acc1714288093e4ace3..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_007000.json +++ /dev/null @@ -1,231 +0,0 @@ -{ - "step": 7000, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 6000, - "val_bpb": 0.9164157458762685, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "wandb_run_id": "d0cda484", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 8352, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "init_from_step": 6000, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok", - "mixture_source_dirs": "{\"midtrain_r30\": \"/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "640d2e56d9ca333095a9e23dcf54d498341552e8", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "branch": { - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_step": 6000, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 140, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 170, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 90, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 8757706752, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 6129975296, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 7881097216, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "43f2fdfbde0b68d4", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 6000, - "config_fingerprint": "43f2fdfbde0b68d4" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48715073, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48715073, - "mixture": { - "cursors": { - "midtrain_r30": { - "file_idx": 10, - "pos": 48715073, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48715073 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 7340163072, - "source_tokens": { - "original": 0, - "midtrain_r30": 1048707072, - "midtrain_r60": 0 - }, - "active_stage_idx": 1 - } - }, - "loop_state": { - "min_val_bpb": 0.9164157458762685, - "smooth_train_loss": 2.6353370917159102, - "total_training_time": 2243.1743581295013, - "stage_start_step": 6000, - "stage_training_flops": 5422802219827200000, - "inherited_parent_flops": 3.25368133189632e+19, - "cumulative_pipeline_training_flops": 3.79596155387904e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_007500.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_007500.json deleted file mode 100644 index 4473a8c6fcd1e158a0836a12b326138f0f82634a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_007500.json +++ /dev/null @@ -1,231 +0,0 @@ -{ - "step": 7500, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 6000, - "val_bpb": 0.9050723402326186, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "wandb_run_id": "d0cda484", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 8352, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "init_from_step": 6000, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok", - "mixture_source_dirs": "{\"midtrain_r30\": \"/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "640d2e56d9ca333095a9e23dcf54d498341552e8", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "branch": { - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_step": 6000, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 140, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 170, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 90, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 8757706752, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 6129975296, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 7881097216, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "43f2fdfbde0b68d4", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 6000, - "config_fingerprint": "43f2fdfbde0b68d4" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 15, - "pos": 73007073, - "epoch": 1, - "pq_idx": 15, - "rg_idx": 73007073, - "mixture": { - "cursors": { - "midtrain_r30": { - "file_idx": 15, - "pos": 73007073, - "epoch": 1, - "pq_idx": 15, - "rg_idx": 73007073 - }, - "midtrain_r60": { - "file_idx": 0, - "pos": 0, - "epoch": 1, - "pq_idx": 0, - "rg_idx": 0 - } - }, - "cumulative_tokens": 7864451072, - "source_tokens": { - "original": 0, - "midtrain_r30": 1572995072, - "midtrain_r60": 0 - }, - "active_stage_idx": 1 - } - }, - "loop_state": { - "min_val_bpb": 0.9050723402326186, - "smooth_train_loss": 2.64315452719698, - "total_training_time": 3379.8451437950134, - "stage_start_step": 6000, - "stage_training_flops": 8134203329740800000, - "inherited_parent_flops": 3.25368133189632e+19, - "cumulative_pipeline_training_flops": 4.0671016648704e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_008000.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_008000.json deleted file mode 100644 index 595b1ca9d831b8eb81c200cc0c4a4764d476d48a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_008000.json +++ /dev/null @@ -1,231 +0,0 @@ -{ - "step": 8000, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 6000, - "val_bpb": 0.890866525025072, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "wandb_run_id": "d0cda484", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 8352, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "init_from_step": 6000, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok", - "mixture_source_dirs": "{\"midtrain_r30\": \"/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "640d2e56d9ca333095a9e23dcf54d498341552e8", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "branch": { - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_step": 6000, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 140, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 170, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 90, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 8757706752, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 6129975296, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 7881097216, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "43f2fdfbde0b68d4", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 6000, - "config_fingerprint": "43f2fdfbde0b68d4" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 7645729, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 7645729, - "mixture": { - "cursors": { - "midtrain_r30": { - "file_idx": 15, - "pos": 89653344, - "epoch": 1, - "pq_idx": 15, - "rg_idx": 89653344 - }, - "midtrain_r60": { - "file_idx": 5, - "pos": 7645729, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 7645729 - } - }, - "cumulative_tokens": 8388739072, - "source_tokens": { - "original": 0, - "midtrain_r30": 1589641216, - "midtrain_r60": 507641856 - }, - "active_stage_idx": 2 - } - }, - "loop_state": { - "min_val_bpb": 0.890866525025072, - "smooth_train_loss": 2.5129958853985905, - "total_training_time": 4517.323269367218, - "stage_start_step": 6000, - "stage_training_flops": 10845604439654400000, - "inherited_parent_flops": 3.25368133189632e+19, - "cumulative_pipeline_training_flops": 4.33824177586176e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_008352.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_008352.json deleted file mode 100644 index 28eba485a72455cdea76a662c5507efdf9c824cd..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/meta_008352.json +++ /dev/null @@ -1,231 +0,0 @@ -{ - "step": 8352, - "training_complete": true, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 6000, - "val_bpb": 0.8864625921976007, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "wandb_run_id": "d0cda484", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8,midtrain-schedule,mix-og,0-30-60", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": 8352, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "init_from_checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "init_from_step": 6000, - "no_init_optimizer": false, - "branch_lr_schedule": "continue", - "branch_parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok", - "mixture_source_dirs": "{\"midtrain_r30\": \"/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok_midtrain_r30\", \"midtrain_r60\": \"/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/pretok_midtrain_r60\"}", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "640d2e56d9ca333095a9e23dcf54d498341552e8", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "branch": { - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_step": 6000, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 140, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 170, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 90, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 8757706752, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 6129975296, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 7881097216, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "43f2fdfbde0b68d4", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 6000, - "config_fingerprint": "43f2fdfbde0b68d4" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 8, - "pos": 76747297, - "epoch": 1, - "pq_idx": 8, - "rg_idx": 76747297, - "mixture": { - "cursors": { - "midtrain_r30": { - "file_idx": 15, - "pos": 89653344, - "epoch": 1, - "pq_idx": 15, - "rg_idx": 89653344 - }, - "midtrain_r60": { - "file_idx": 8, - "pos": 76747297, - "epoch": 1, - "pq_idx": 8, - "rg_idx": 76747297 - } - }, - "cumulative_tokens": 8757837824, - "source_tokens": { - "original": 0, - "midtrain_r30": 1589641216, - "midtrain_r60": 876740608 - }, - "active_stage_idx": 2 - } - }, - "loop_state": { - "min_val_bpb": 0.8864625921976007, - "smooth_train_loss": 2.446982447319594, - "total_training_time": 5318.38954949379, - "stage_start_step": 6000, - "stage_training_flops": 12754430821033574400, - "inherited_parent_flops": 3.25368133189632e+19, - "cumulative_pipeline_training_flops": 4.5291244139996774e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_006500.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_006500.pt deleted file mode 100644 index 7d64b883be4b64dda220575ef9f19f397026a9f2..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_006500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9b164ffcc7be627e3afb0b0e8ae0a5ff3b97a46f8aa01c9d37f3b9c3b004b572 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_007000.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_007000.pt deleted file mode 100644 index ec2998bc6ed84a31b1f9e60ba00615098e9ea47e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_007000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9fe7a666b22db61b0eb89da7b07ea90fd606b31fdbe6e2d665cea4e09a54dd40 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_007500.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_007500.pt deleted file mode 100644 index af3763aa61cb6f03f7984e8b482804688d85219d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_007500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0abeff0941358f864da576247a97de9d9c7ae4d00e5e4e08115b32fe0408bdeb -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_008000.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_008000.pt deleted file mode 100644 index a5c8f847a906799ccb3dd644b1b4a536ae41b72e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_008000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0950335e48894bc28488bab0a27c7bae5fa49389290a729216f12da88db49658 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_008352.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_008352.pt deleted file mode 100644 index 4a5eab61729e3603eb233898880161ff421c8366..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/model_008352.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:dd2e57b069247cb6eaebbbf6e8acc584b8b8d94d30910bc7cd989210424af1a1 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_006500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_006500_rank0.pt deleted file mode 100644 index 5405851816cb69929b28cd77276c22b11cf05200..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_006500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8222dfefc5520aaf6cb5914daf96f97bc86b176bd055c9d737890e84c9e960ef -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_006500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_006500_rank1.pt deleted file mode 100644 index 0435772226add041a3b5ca9178e66809e8085bff..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_006500_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d11a80ba154d4927cacd66ddc6059fc6408944bf484fc1a37536db6ef974dfce -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_006500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_006500_rank2.pt deleted file mode 100644 index 27675a76e3d72e3c8dd877862ab12f7cf47a8e6e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_006500_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:15ab49eacd43bbc4f3c9949fcd37cd84118aedf94a8449bdf4536bd280f540d0 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_006500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_006500_rank3.pt deleted file mode 100644 index b8ee303b71378e7eb49a995c20d4d8a93283b16a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_006500_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:68317bde289fc05e38737cab8ed897f0fe3555da32ac1d45ae947995aff526ee -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007000_rank0.pt deleted file mode 100644 index f3e833844022d495cce35a2e36016ecead56b791..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9d8962c430aa9630a40edb89836363111497e56b521bed8290770d242d86e576 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007000_rank1.pt deleted file mode 100644 index 545bbd7ae49b33e99ea0b3862025ca23d67d4e30..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007000_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:922d166bb739a0b321085bec48de5f1ff1bf6e58ac805ad50dd4ec59282e905f -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007000_rank2.pt deleted file mode 100644 index af05d9a7175251eca64dc5de87c376085a5b51ca..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007000_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e421c62a57d07c517a6365e31cbdfa87806c9669a25bc12afb19110db86af537 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007000_rank3.pt deleted file mode 100644 index 0b2ee4404eedd6f20fdd4e63cf2875257aef70d9..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007000_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3676560bd5ee261a1d30ef5b0a0309827486d043b527de6989327bd8b5b45421 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007500_rank0.pt deleted file mode 100644 index 11f2339a00671704b5ad3c619840db7406da573f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b25117908bf46c2401d78a8a623ee44e859c4b6af761bcaa9daf98e73c08c9e9 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007500_rank1.pt deleted file mode 100644 index 931943c3f119d407074261851836f438e13e8ac7..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007500_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e1f56c152f095b226b639b11dc2ec0ac95ba5a7601f4493f6375dc5083c6729e -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007500_rank2.pt deleted file mode 100644 index c5e48632468fd20851d558886fc033474ab54cbf..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007500_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d64e27870d1701eeeb5e6ccf41524b06eba86ecb1df13683ff65a02287bf684f -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007500_rank3.pt deleted file mode 100644 index 9bdedba4c705f4f5a4498fb4456029a2d27f6a4d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_007500_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:98925db0bfd97e4c45e5bc2d6d2dbaca5eabdbbb12a307eed6080311d70d7e05 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008000_rank0.pt deleted file mode 100644 index 668ab4d9a6d88595f1fedacc8c769c7700c87733..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4a85181a8513387352fbb089dc17bf8828d1d4ebac47c49d173605edf1a99fe8 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008000_rank1.pt deleted file mode 100644 index 6edad79d101e0cf0ada0072d715a47b6f487083d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008000_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:de5c5e81af359caf49287a7f5cf812ed31565adfc9c2e3d13c793654368dd20a -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008000_rank2.pt deleted file mode 100644 index a08560632eefa985e2016094465bf9d22301601d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008000_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:ae5e83b2b872caf16699e319fe8e79d1edeeade57e087446667530bf2573ae40 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008000_rank3.pt deleted file mode 100644 index d8dd26146feaadb03d14f20f3c3480eea6c867f0..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008000_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc19e7de51b183fe37e870d41f54ac4bd0916cea69427867795d4e7dd77dc3e9 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008352_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008352_rank0.pt deleted file mode 100644 index ab66e9738a2332cd155d820800f211d34dad6ed6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008352_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:da8d750253b41f3f4e88db1f7a81d864f3a6c8704503860e5878a4af78a4b91a -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008352_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008352_rank1.pt deleted file mode 100644 index ee2ccb0f2a8baf0464aba69616470864efcef14b..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008352_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:31abfbbfdf30576fc9cf8be6f4b707ee9770806a4cee172d62128a60f03ee8b8 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008352_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008352_rank2.pt deleted file mode 100644 index 588950580579d1a8ec53748470392943976314dc..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008352_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e35d122f69594d1606daf881332ec492fd5cf7b70f63ab0ea0f2530222b1c9c9 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008352_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008352_rank3.pt deleted file mode 100644 index ea767da291edb82174cce399e8b5fa3a6fd83719..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/base_checkpoints/optim_008352_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:253eeb5478238b4f3df442186ca5518d99cb8089082e3b79b70c95b95c30adcb -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/config.json deleted file mode 100644 index 0e95e6c6abcca3d11902c247a39300f6e852ced7..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/config.json +++ /dev/null @@ -1,114 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "branch": { - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_step": 6000, - "lr_schedule": "continue", - "load_optimizer": true - }, - "datasets": { - "original": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 140, - "download_workers": 4 - }, - "midtrain_r30": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_30/data", - "validation_shard": 1262, - "num_train_shards": 170, - "download_workers": 4 - }, - "midtrain_r60": { - "adapter": "parquet_shards", - "repo": "zachnorton03/think-midtrain", - "revision": "main", - "subfolder": "mixed/ratio_60/data", - "validation_shard": 523, - "num_train_shards": 90, - "download_workers": 4 - } - }, - "mixture_schedule": { - "total_tokens": 8757706752, - "seed_data": 1234, - "max_epochs": 4, - "stages": [ - { - "name": "base", - "start_tokens": 0, - "source": "original" - }, - { - "name": "injection", - "start_tokens": 6129975296, - "source": "midtrain_r30" - }, - { - "name": "decay_mix", - "start_tokens": 7881097216, - "source": "midtrain_r60" - } - ] - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8", - "midtrain-schedule", - "mix-og", - "0-30-60" - ] - }, - "config_fingerprint": "43f2fdfbde0b68d4", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og" -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/evals/core.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/evals/core.json deleted file mode 100644 index 8163192ced4098cd8ac479fd15b5d479009d7703..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 8352)", - "step": 8352, - "bpb": {}, - "core_metric": 0.14018200582682225, - "core_results": { - "hellaswag_zeroshot": 0.33041226863861084, - "jeopardy": 0.012281530536711216, - "bigbench_qa_wikidata": 0.24049012362957, - "arc_easy": 0.41456228494644165, - "arc_challenge": 0.239761084318161, - "copa": 0.5999999642372131, - "commonsense_qa": 0.32514333724975586, - "piqa": 0.5854189395904541, - "openbook_qa": 0.26200002431869507, - "lambada_openai": 0.3720162808895111, - "hellaswag": 0.33379805088043213, - "winograd": 0.6410256624221802, - "winogrande": 0.5177584886550903, - "bigbench_dyck_languages": 0.13700000941753387, - "agi_eval_lsat_ar": 0.2652173936367035, - "bigbench_cs_algorithms": 0.396212100982666, - "bigbench_operators": 0.10476190596818924, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.14124882221221924, - "coqa": 0.18476763367652893, - "boolq": 0.5993883609771729, - "bigbench_language_identification": 0.25679999589920044 - }, - "centered_results": { - "hellaswag_zeroshot": 0.10721635818481445, - "jeopardy": 0.012281530536711216, - "bigbench_qa_wikidata": 0.24049012362957, - "arc_easy": 0.21941637992858887, - "arc_challenge": -0.013651887575785318, - "copa": 0.19999992847442627, - "commonsense_qa": 0.1564291715621948, - "piqa": 0.1708378791809082, - "openbook_qa": 0.016000032424926758, - "lambada_openai": 0.3720162808895111, - "hellaswag": 0.11173073450724284, - "winograd": 0.28205132484436035, - "winogrande": 0.035516977310180664, - "bigbench_dyck_languages": 0.13700000941753387, - "agi_eval_lsat_ar": 0.08152174204587935, - "bigbench_cs_algorithms": 0.396212100982666, - "bigbench_operators": 0.10476190596818924, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.14124882221221924, - "coqa": 0.18476763367652893, - "boolq": -0.054241155323229324, - "bigbench_language_identification": 0.18239823531265176 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/evals/samples.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/evals/samples.json deleted file mode 100644 index a6a2dcea808b8ffe4ac471142bad747cfa64fc0d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 8352)", - "step": 8352, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of France.\n\nThe capital of France is the capital of France.\n\nThe" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is\n\nA. 1. 2. 3. 4. " - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Sunday.\n\nThe Sunday after the Friday, the Sunday after the Sunday, the Sunday" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is cold, and vice versa.\n\nThe opposite of cold is hot, and vice versa" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: Mercury, Venus, Mars, Jupiter, Saturn, Uranus, Neptune," - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a deep, rich, velvety crimson, with a large, full," - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times 5 is contained in 13, and 3 is" - } - ], - "unconditioned_samples": [ - "<|bos|>per anus ullams v .. niteen erl\n\nSAC OF THE UNKNOWN MEASURES\n\nJustifiable Xiwerermarg \u00abnsng van Fairie half Psos 1n FSA || Y | ||| -_|||| || | || || | |||_ | | | || | ||| |}| |_|_|_| || |_|| |) | || | | | | | _| ||| | |\n\nai i Vi : } J | | |", - "<|bos|>Lady Encurt, bi le Scotch wu cften mud sbee, for Lady Percrash Claveringastes forlorn she hard, fine or dime he ran ahead of work, veile, tat Nells under the wualdkhorns wan garee mad dulep, Woman Asking, he one see, weeded som dulep a way South Unionalysnowed bruts arayn made him kyill, sic lass tat Brappy Beuld sco declared review of rory allula idiotoo necessary, all loues, Na-momber), over wu my", - "<|bos|>!\n\nIt is the coss ism of the brute Dy-a-sms him h-koo-stum\"- Vanderplaste\n\nBoulde Pdst F who is he who is idist ode a china\n\nApril 100 Jy. 52.. cison had tried ad all the ships *last Spring\n\nIf he reported errator life ament's' and had but a few chances but formed Hands, been themost of the bunch rrived nt wrecks eloped, way became united, Mou, form, where the rriend of cancel that He would get entitled to\n\nPove was the", - "<|bos|> ministered by 350 ww Maintaine It IS CHA ehweect end \u201e concited vie BS A TAM NgS\n\nMATT. XX. 20.] 81\n\nPaul; and the occasion of this journey was the continuance of the schism between the Paulicians and the Catholic Christians. He gives an account of this persecution which he had first received from Peter, and which, as being handed down from the days of the apostles and the martyrs, was handed on to him by tradition, and would consequently be remembered by infant children. For some time, this allusion to Sunday broke through the Established religion", - "<|bos|>D\n\nBRARY COLLEGE\n\nEach separate hearth has its own peculiar i\n\nCONTRACT XXII\n\nW.L Watts\n\nLEARNINGS.\n\nRetaining fee (Workshop fee (Sheld-iron fee)\n\nCheap fee (Washboard fee)\n\nBox fee (Manual bench fee)\n\nGrand total (Manual bench fee)\n\nThat is the sole profit we have to prize.*\n\n4. What are examples?\n\nCHAPTER VIII.\n\nFEE SIMPLE RENT CHAPTER IX.\n\nCONTRACT SECTION SECTION SECTION\n\n1. The value of such a lease is Its perpetuity\n\n2. How the deed of lease was reformed\n\n3. The", - "<|bos|>There is a quiet play-place. when pictures and and was being finished, BURWILL, the architect, of late years.\n\nHARDY, now of S.C.R., has come up from London to study it by the way.\n\nHARDY has just been having some talk with HARRIET BOLTON, sister of Marshman, one of his assistants and \"pigs,\" for an opportunity of getting some arrangement made for her, and she shakes her head.\n\nHARDY and MARSTON have some kind of an understanding with CAIUS ELMER ELLIS.\n\nELLIS enters, and the two rise to salute", - "<|bos|>iman have published,\n\nTo Architects, etc., Manchester, Eng.\n\n39-42 S. MARLITT TRUS B\n\nAl7@\n\nA 180-32 3.5 Ppela\n\nF37. Reduced Facsimile of Batrachia (FERRARA\n\n79--82 VASARA)\n\nG :\n\n23!uuefn of\n\nBULLETIN\n\nnf\n\nCOLLECTION\n\n! : =e\n\nqe\n\nae\n\n--e\n\nvbs mgest\n\ny fine &\n\nAT\n\nb B Fd Se Sviar (nai yy E", - "<|bos|>MY Dogs Mendingable and Foxy\n\nYou Matty Neal toning the tone\n\nCh'illie Row, Sntis ^ large'\n\nJ-onlir 1926\n\nJanuary 30, 1926\n\nI fully appreciate the kindness that has been shown in naming me, and I regret having been unable to have a complete hunting record made for some years.\n\nRuppapon has _been my Hubel friend for many years. Harris gives to me. every wild trip about the Department. Wiap is my friend, and all with me are equally willing. If you, dear Rupp" - ] -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/evals/val_bpb.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/evals/val_bpb.json deleted file mode 100644 index b68f63c136c0e27e2485d010572e2e7f2cbbaa52..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 8352)", - "step": 8352, - "bpb": { - "val": 0.8601897122100801 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/run.json deleted file mode 100644 index c075769f4818634153a3bb556da041f221bceb2d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 6000, - "branch_parent_step": 6000, - "config_fingerprint": "43f2fdfbde0b68d4", - "wandb_run_id": "d0cda484", - "created_at": 1785617422 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/summary.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/summary.json deleted file mode 100644 index 79ae107e00e544c6ded8384d38ee739c34ba9c21..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/summary.json +++ /dev/null @@ -1,95 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 6000, - "dataset": "zachnorton03/think-midtrain", - "dataset_revision": "main", - "step": 8352, - "depth": 24, - "target_param_data_ratio": 12, - "training_tokens": 8757706752, - "final_sampled_val_bpb": 0.8864625921976007, - "minimum_sampled_val_bpb": 0.8864625921976007, - "full_val_bpb": 0.8601897122100801, - "core_metric": 0.14018200582682225, - "centered_results": { - "hellaswag_zeroshot": 0.10721635818481445, - "jeopardy": 0.012281530536711216, - "bigbench_qa_wikidata": 0.24049012362957, - "arc_easy": 0.21941637992858887, - "arc_challenge": -0.013651887575785318, - "copa": 0.19999992847442627, - "commonsense_qa": 0.1564291715621948, - "piqa": 0.1708378791809082, - "openbook_qa": 0.016000032424926758, - "lambada_openai": 0.3720162808895111, - "hellaswag": 0.11173073450724284, - "winograd": 0.28205132484436035, - "winogrande": 0.035516977310180664, - "bigbench_dyck_languages": 0.13700000941753387, - "agi_eval_lsat_ar": 0.08152174204587935, - "bigbench_cs_algorithms": 0.396212100982666, - "bigbench_operators": 0.10476190596818924, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.14124882221221924, - "coqa": 0.18476763367652893, - "boolq": -0.054241155323229324, - "bigbench_language_identification": 0.18239823531265176 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of France.\n\nThe capital of France is the capital of France.\n\nThe" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is\n\nA. 1. 2. 3. 4. " - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Sunday.\n\nThe Sunday after the Friday, the Sunday after the Sunday, the Sunday" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is cold, and vice versa.\n\nThe opposite of cold is hot, and vice versa" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: Mercury, Venus, Mars, Jupiter, Saturn, Uranus, Neptune," - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a deep, rich, velvety crimson, with a large, full," - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times 5 is contained in 13, and 3 is" - } - ], - "unconditioned_samples": [ - "<|bos|>per anus ullams v .. niteen erl\n\nSAC OF THE UNKNOWN MEASURES\n\nJustifiable Xiwerermarg \u00abnsng van Fairie half Psos 1n FSA || Y | ||| -_|||| || | || || | |||_ | | | || | ||| |}| |_|_|_| || |_|| |) | || | | | | | _| ||| | |\n\nai i Vi : } J | | |", - "<|bos|>Lady Encurt, bi le Scotch wu cften mud sbee, for Lady Percrash Claveringastes forlorn she hard, fine or dime he ran ahead of work, veile, tat Nells under the wualdkhorns wan garee mad dulep, Woman Asking, he one see, weeded som dulep a way South Unionalysnowed bruts arayn made him kyill, sic lass tat Brappy Beuld sco declared review of rory allula idiotoo necessary, all loues, Na-momber), over wu my", - "<|bos|>!\n\nIt is the coss ism of the brute Dy-a-sms him h-koo-stum\"- Vanderplaste\n\nBoulde Pdst F who is he who is idist ode a china\n\nApril 100 Jy. 52.. cison had tried ad all the ships *last Spring\n\nIf he reported errator life ament's' and had but a few chances but formed Hands, been themost of the bunch rrived nt wrecks eloped, way became united, Mou, form, where the rriend of cancel that He would get entitled to\n\nPove was the", - "<|bos|> ministered by 350 ww Maintaine It IS CHA ehweect end \u201e concited vie BS A TAM NgS\n\nMATT. XX. 20.] 81\n\nPaul; and the occasion of this journey was the continuance of the schism between the Paulicians and the Catholic Christians. He gives an account of this persecution which he had first received from Peter, and which, as being handed down from the days of the apostles and the martyrs, was handed on to him by tradition, and would consequently be remembered by infant children. For some time, this allusion to Sunday broke through the Established religion", - "<|bos|>D\n\nBRARY COLLEGE\n\nEach separate hearth has its own peculiar i\n\nCONTRACT XXII\n\nW.L Watts\n\nLEARNINGS.\n\nRetaining fee (Workshop fee (Sheld-iron fee)\n\nCheap fee (Washboard fee)\n\nBox fee (Manual bench fee)\n\nGrand total (Manual bench fee)\n\nThat is the sole profit we have to prize.*\n\n4. What are examples?\n\nCHAPTER VIII.\n\nFEE SIMPLE RENT CHAPTER IX.\n\nCONTRACT SECTION SECTION SECTION\n\n1. The value of such a lease is Its perpetuity\n\n2. How the deed of lease was reformed\n\n3. The", - "<|bos|>There is a quiet play-place. when pictures and and was being finished, BURWILL, the architect, of late years.\n\nHARDY, now of S.C.R., has come up from London to study it by the way.\n\nHARDY has just been having some talk with HARRIET BOLTON, sister of Marshman, one of his assistants and \"pigs,\" for an opportunity of getting some arrangement made for her, and she shakes her head.\n\nHARDY and MARSTON have some kind of an understanding with CAIUS ELMER ELLIS.\n\nELLIS enters, and the two rise to salute", - "<|bos|>iman have published,\n\nTo Architects, etc., Manchester, Eng.\n\n39-42 S. MARLITT TRUS B\n\nAl7@\n\nA 180-32 3.5 Ppela\n\nF37. Reduced Facsimile of Batrachia (FERRARA\n\n79--82 VASARA)\n\nG :\n\n23!uuefn of\n\nBULLETIN\n\nnf\n\nCOLLECTION\n\n! : =e\n\nqe\n\nae\n\n--e\n\nvbs mgest\n\ny fine &\n\nAT\n\nb B Fd Se Sviar (nai yy E", - "<|bos|>MY Dogs Mendingable and Foxy\n\nYou Matty Neal toning the tone\n\nCh'illie Row, Sntis ^ large'\n\nJ-onlir 1926\n\nJanuary 30, 1926\n\nI fully appreciate the kindness that has been shown in naming me, and I regret having been unable to have a complete hunting record made for some years.\n\nRuppapon has _been my Hubel friend for many years. Harris gives to me. every wild trip about the Department. Wiap is my friend, and all with me are equally willing. If you, dear Rupp" - ], - "training_time_seconds": 5318.38954949379, - "stage_training_flops": 1.2754430821033574e+19, - "inherited_parent_flops": 3.25368133189632e+19, - "cumulative_pipeline_training_flops": 4.5291244139996774e+19, - "config_fingerprint": "43f2fdfbde0b68d4", - "git_commit_sha": "640d2e56d9ca333095a9e23dcf54d498341552e8", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/d0cda484", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og", - "dataset_fingerprint": "4fdac5e7831c7e7a", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "branch_parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "branch_parent_step": 6000, - "branch_lr_schedule": "continue", - "stage_training_tokens": 2466250752 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer/token_bytes.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer/tokenizer.pkl b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-mix-og/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-anneal-ratio0-v1/train_stdouterr.log b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-anneal-ratio0-v1/train_stdouterr.log deleted file mode 100644 index 1ebf191b3b29991e59d0a26ab2047bfb48ee51fc..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-anneal-ratio0-v1/train_stdouterr.log +++ /dev/null @@ -1,206 +0,0 @@ - - █████ █████ - ░░███ ░░███ - ████████ ██████ ████████ ██████ ██████ ░███████ ██████ ███████ - ░░███░░███ ░░░░░███ ░░███░░███ ███░░███ ███░░███ ░███░░███ ░░░░░███░░░███░ - ░███ ░███ ███████ ░███ ░███ ░███ ░███░███ ░░░ ░███ ░███ ███████ ░███ - ░███ ░███ ███░░███ ░███ ░███ ░███ ░███░███ ███ ░███ ░███ ███░░███ ░███ ███ - ████ █████░░████████ ████ █████░░██████ ░░██████ ████ █████░░███████ ░░█████ - ░░░░ ░░░░░ ░░░░░░░░ ░░░░ ░░░░░ ░░░░░░ ░░░░░░ ░░░░ ░░░░░ ░░░░░░░░ ░░░░░ - -Autodetected device type: cuda -2026-07-22 19:24:22,223 - nanochat.common - INFO - Distributed world size: 1 -GPU: NVIDIA A100-SXM4-80GB | Peak FLOPS (BF16): 3.12e+14 -COMPUTE_DTYPE: torch.bfloat16 (auto-detected: CUDA SM 80 (bf16 supported)) -wandb: [wandb.login()] Loaded credentials for https://api.wandb.ai from WANDB_API_KEY. -wandb: Currently logged in as: owensvoorhees (jbduran-thinkingmachinesncsu) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin -wandb: Using an existing wandb-core service via WANDB_SERVICE. -wandb: setting up run fcb6fc04 -wandb: Tracking run with wandb version 0.28.0 -wandb: Run data is saved locally in /content/think.nano/wandb/run-20260722_192422-fcb6fc04 -wandb: Run `wandb offline` to turn off syncing. -wandb: Syncing run clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-anneal-ratio0-v1 -wandb: ⭐️ View project at https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano -wandb: 🚀 View run at https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/fcb6fc04 -!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!! -WARNING: Flash Attention 3 not available, using PyTorch SDPA fallback -WARNING: Training will be less efficient without FA3 -WARNING: SDPA has no support for sliding window attention (window_pattern='SSSL'). Your GPU utilization will be terrible. -WARNING: Recommend using --window-pattern L for full context attention without alternating sliding window patterns. -!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!! -Vocab size: 32,768 -Model config: -{ - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" -} -Resuming optimization from step 5500 -Parameter counts: -wte : 50,331,648 -value_embeds : 603,979,776 -lm_head : 50,331,648 -transformer_matrices : 679,478,976 -scalars : 74 -total : 1,384,122,122 -Estimated FLOPs per token: 5.171587e+09 -Scaling LRs by 1.4142 for batch size 1,048,576 (reference: 524,288) -Scaling weight decay from 0.280000 to 0.059738 for depth 24 -Scaling the LR for the AdamW parameters ∝1/√(1536/768) = 0.707107 -Using pretokenized uint16 dataloader: /content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-anneal-ratio0-v1/pretok -Using user-provided number of iterations: 5,750 -Total number of training tokens: 6,029,312,000 -Tokens : Scaling params ratio: 8.26 -Total training FLOPs estimate: 3.118111e+19 -Tokens / micro-batch / rank: 8 x 4096 = 32,768 -Tokens / micro-batch: 32,768 -Total batch size 1,048,576 => gradient accumulation steps: 32 -Step 05500 | Validation bpb: 0.816509 -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] failed while attempting to run meta for aten.lerp_.Tensor -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] Traceback (most recent call last): -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] File "/usr/local/lib/python3.12/dist-packages/torch/_subclasses/fake_tensor.py", line 2904, in _dispatch_impl -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] r = func(*args, **kwargs) -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] ^^^^^^^^^^^^^^^^^^^^^ -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] File "/usr/local/lib/python3.12/dist-packages/torch/_ops.py", line 865, in __call__ -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] return self._op(*args, **kwargs) -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^^^ -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] File "/usr/local/lib/python3.12/dist-packages/torch/_meta_registrations.py", line 8652, in _fn -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] out = fn(self, *args, **kwargs) -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^^^ -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] File "/usr/local/lib/python3.12/dist-packages/torch/_ops.py", line 1269, in __call__ -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] return self._op(*args, **kwargs) -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^^^ -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] File "/usr/local/lib/python3.12/dist-packages/torch/_prims_common/wrappers.py", line 314, in _fn -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] result = fn(*args, **kwargs) -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] ^^^^^^^^^^^^^^^^^^^ -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] File "/usr/local/lib/python3.12/dist-packages/torch/_meta_registrations.py", line 8678, in lerp -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] return elementwise_meta( -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] ^^^^^^^^^^^^^^^^^ -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] File "/usr/local/lib/python3.12/dist-packages/torch/_meta_registrations.py", line 91, in elementwise_meta -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] args = _maybe_broadcast(*args) -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^ -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] File "/usr/local/lib/python3.12/dist-packages/torch/_refs/__init__.py", line 471, in _maybe_broadcast -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] common_shape = _broadcast_shapes( -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] ^^^^^^^^^^^^^^^^^^ -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] File "/usr/local/lib/python3.12/dist-packages/torch/_refs/__init__.py", line 459, in _broadcast_shapes -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] torch._check( -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] File "/usr/local/lib/python3.12/dist-packages/torch/__init__.py", line 1753, in _check -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] _check_with(RuntimeError, cond, message) # pyrefly: ignore [bad-argument-type] -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] File "/usr/local/lib/python3.12/dist-packages/torch/__init__.py", line 1735, in _check_with -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] raise error_type(message_evaluated) -E0722 19:28:07.170000 10035 torch/_subclasses/fake_tensor.py:2908] [1/0] RuntimeError: Attempting to broadcast a dimension of length 32768 at -2! Mismatching argument at index 1 had torch.Size([32768, 1536]); but expected shape should be broadcastable to [8192, 1536] -Traceback (most recent call last): - File "", line 198, in _run_module_as_main - File "", line 88, in _run_code - File "/content/think.nano/scripts/base_train.py", line 611, in - optimizer.step() - File "/usr/local/lib/python3.12/dist-packages/torch/optim/optimizer.py", line 533, in wrapper - out = func(*args, **kwargs) - ^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/utils/_contextlib.py", line 124, in decorate_context - return func(*args, **kwargs) - ^^^^^^^^^^^^^^^^^^^^^ - File "/content/think.nano/nanochat/optim.py", line 307, in step - self._step_adamw(group) - File "/content/think.nano/nanochat/optim.py", line 243, in _step_adamw - adamw_step_fused( - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/eval_frame.py", line 1024, in compile_wrapper - return fn(*args, **kwargs) - ^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/convert_frame.py", line 2316, in __call__ - result = self._torchdynamo_orig_backend( - ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/convert_frame.py", line 729, in __call__ - result = _compile( - ^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/convert_frame.py", line 1827, in _compile - guarded_code, tracer_output = compile_inner(code, one_graph, hooks) - ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_utils_internal.py", line 96, in wrapper_function - return function(*args, **kwargs) - ^^^^^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/convert_frame.py", line 1500, in compile_inner - return _compile_inner(code, one_graph, hooks) - ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/convert_frame.py", line 1534, in _compile_inner - dynamo_output = compile_frame( - ^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/convert_frame.py", line 1408, in compile_frame - bytecode, tracer_output = transform_code_object(code, transform) - ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/bytecode_transformation.py", line 1608, in transform_code_object - tracer_output = transformations(instructions, code_options) - ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/convert_frame.py", line 1380, in transform - tracer_output = trace_frame( - ^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/convert_frame.py", line 341, in _fn - return fn(*args, **kwargs) - ^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/convert_frame.py", line 863, in trace_frame - run_tracer() - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/convert_frame.py", line 844, in run_tracer - tracer.run() - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/symbolic_convert.py", line 1794, in run - while self.step(): - ^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/symbolic_convert.py", line 1459, in step - self.dispatch_table[inst.opcode](self, inst) - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/symbolic_convert.py", line 1001, in wrapper - return inner_fn(self, inst) - ^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/symbolic_convert.py", line 4093, in CALL - self._call(inst) - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/symbolic_convert.py", line 4084, in _call - self.call_function(fn, args, kwargs) - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/symbolic_convert.py", line 1360, in call_function - self.push(fn.call_function(self, args, kwargs)) # type: ignore[arg-type] - ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/variables/misc.py", line 1340, in call_function - return self.obj.call_method(tx, self.name, list(args), kwargs) - ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/variables/tensor.py", line 850, in call_method - result = wrap_fx_proxy(tx, proxy) - ^^^^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/variables/builder.py", line 3032, in wrap_fx_proxy - return wrap_fx_proxy_cls(target_cls=TensorVariable, **kwargs) - ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/variables/builder.py", line 3107, in wrap_fx_proxy_cls - out: VTTypeAlias = _wrap_fx_proxy( - ^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/variables/builder.py", line 3231, in _wrap_fx_proxy - example_value = get_fake_value(proxy.node, tx, allow_non_graph_fake=True) - ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/utils.py", line 3627, in get_fake_value - return _get_fake_value_impl(node, tx, allow_non_graph_fake) - ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/utils.py", line 3818, in _get_fake_value_impl - _wrap_graph_break_with_torch_runtime_err( - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/utils.py", line 3616, in _wrap_graph_break_with_torch_runtime_err - raise exc.with_traceback(e.__traceback__) from None - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/utils.py", line 3613, in _wrap_graph_break_with_torch_runtime_err - gb_fn() - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/utils.py", line 3819, in - lambda: unimplemented( - ^^^^^^^^^^^^^^ - File "/usr/local/lib/python3.12/dist-packages/torch/_dynamo/exc.py", line 631, in unimplemented - raise Unsupported( -torch._dynamo.exc.TorchRuntimeError: RuntimeError when making fake tensor call - Explanation: Dynamo failed to run FX node with fake tensors: call_method lerp_(*(FakeTensor(..., device='cuda:0', size=(8192, 1536)), FakeTensor(..., device='cuda:0', size=(32768, 1536)), FakeTensor(..., size=())), **{}): got RuntimeError('Attempting to broadcast a dimension of length 32768 at -2! Mismatching argument at index 1 had torch.Size([32768, 1536]); but expected shape should be broadcastable to [8192, 1536]') - Hint: Your code may result in an error when running in eager. Please double check that your code doesn't contain a similar error when actually running eager/uncompiled. You can do this by removing the `torch.compile` call, or by using `torch.compiler.set_stance("force_eager")`. - - Developer debug context: - - For more details about this graph break, please visit: https://meta-pytorch.github.io/compile-graph-break-site/gb/gb4315.html - -from user code: - File "/content/think.nano/nanochat/optim.py", line 42, in adamw_step_fused - exp_avg.lerp_(grad, 1 - beta1_t) - -Set TORCHDYNAMO_VERBOSE=1 for the internal stack trace (please do this especially if you're reporting a bug to PyTorch). For even more developer context, set TORCH_LOGS="+dynamo" - diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_000500.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_000500.json deleted file mode 100644 index 5fb750e5aee183b5df2189227d89f0c24298fd7e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 1.1953384268774052, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24324769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24324769 - }, - "loop_state": { - "min_val_bpb": 1.1953384268774052, - "smooth_train_loss": 3.5724794941151132, - "total_training_time": 1093.107646226883, - "stage_training_flops": 2711401109913600000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2711401109913600000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_001000.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_001000.json deleted file mode 100644 index f7e574426564f9e244fa48d64c9d936c0d671c39..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 1.1077649147516626, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48616769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48616769 - }, - "loop_state": { - "min_val_bpb": 1.1077649147516626, - "smooth_train_loss": 3.1062403539181314, - "total_training_time": 2205.066916704178, - "stage_training_flops": 5422802219827200000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 5422802219827200000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_001500.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_001500.json deleted file mode 100644 index 4ef84b6c45e667d68781cadccca463e66b86e1e5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 1.084373843120419, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 15, - "pos": 72908769, - "epoch": 1, - "pq_idx": 15, - "rg_idx": 72908769 - }, - "loop_state": { - "min_val_bpb": 1.084373843120419, - "smooth_train_loss": 3.195503519346087, - "total_training_time": 3315.8829300403595, - "stage_training_flops": 8134203329740800000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 8134203329740800000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_002000.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_002000.json deleted file mode 100644 index dbfb1aa7055039f8898bc53b6205965805f773df..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 1.0782916447573232, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 20, - "pos": 97200769, - "epoch": 1, - "pq_idx": 20, - "rg_idx": 97200769 - }, - "loop_state": { - "min_val_bpb": 1.0657483238795336, - "smooth_train_loss": 3.0703979980689327, - "total_training_time": 4422.268032073975, - "stage_training_flops": 10845604439654400000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 10845604439654400000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_002500.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_002500.json deleted file mode 100644 index 7231595c8f0b6a7460885120106c2c398a6e4e39..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_002500.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 2500, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 1.0526218621363435, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 26, - "pos": 21492769, - "epoch": 1, - "pq_idx": 26, - "rg_idx": 21492769 - }, - "loop_state": { - "min_val_bpb": 1.0526218621363435, - "smooth_train_loss": 3.1037243528968625, - "total_training_time": 5532.842117547989, - "stage_training_flops": 13557005549568000000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 13557005549568000000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_003000.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_003000.json deleted file mode 100644 index da318d62c57a70c7a28bb3487618f0a04cdacb60..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_003000.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 3000, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 1.0427625547182284, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 31, - "pos": 45784769, - "epoch": 1, - "pq_idx": 31, - "rg_idx": 45784769 - }, - "loop_state": { - "min_val_bpb": 1.0411011903064002, - "smooth_train_loss": 2.851544385420298, - "total_training_time": 6638.3725233078, - "stage_training_flops": 16268406659481600000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 16268406659481600000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_003500.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_003500.json deleted file mode 100644 index c4c69dad0cf5bf673dcdde43660e8aa67f8c5acc..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_003500.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 3500, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 1.0195827444555374, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 36, - "pos": 70076769, - "epoch": 1, - "pq_idx": 36, - "rg_idx": 70076769 - }, - "loop_state": { - "min_val_bpb": 1.0195827444555374, - "smooth_train_loss": 2.756684128888915, - "total_training_time": 7743.951657295227, - "stage_training_flops": 18979807769395200000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 18979807769395200000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_004000.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_004000.json deleted file mode 100644 index f6638bbac9c88043a0490131cc33f79ef8bfb72f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_004000.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 4000, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 1.0052481244227562, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 41, - "pos": 94368769, - "epoch": 1, - "pq_idx": 41, - "rg_idx": 94368769 - }, - "loop_state": { - "min_val_bpb": 1.0052481244227562, - "smooth_train_loss": 2.905858668666241, - "total_training_time": 8849.971616983414, - "stage_training_flops": 21691208879308800000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 21691208879308800000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_004500.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_004500.json deleted file mode 100644 index f2a5327735ce6ed0d9b2c0b4ca434a2831395bef..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_004500.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 4500, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 0.9728434170391407, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 47, - "pos": 18660769, - "epoch": 1, - "pq_idx": 47, - "rg_idx": 18660769 - }, - "loop_state": { - "min_val_bpb": 0.9728434170391407, - "smooth_train_loss": 2.609150689793686, - "total_training_time": 9956.485046625137, - "stage_training_flops": 24402609989222400000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 24402609989222400000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_005000.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_005000.json deleted file mode 100644 index b6bb716c11ae2585bace35a47dd279ef7ad1a128..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_005000.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 5000, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 0.9678782660975384, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 52, - "pos": 42952769, - "epoch": 1, - "pq_idx": 52, - "rg_idx": 42952769 - }, - "loop_state": { - "min_val_bpb": 0.9678782660975384, - "smooth_train_loss": 2.770639739162406, - "total_training_time": 11063.301546573639, - "stage_training_flops": 27114011099136000000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 27114011099136000000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_005500.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_005500.json deleted file mode 100644 index ce3104ab029166c63b50833afb63e89b34c2d974..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_005500.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 5500, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 0.9620444309513437, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 57, - "pos": 67244769, - "epoch": 1, - "pq_idx": 57, - "rg_idx": 67244769 - }, - "loop_state": { - "min_val_bpb": 0.9620444309513437, - "smooth_train_loss": 2.7810625137611145, - "total_training_time": 12169.80887722969, - "stage_training_flops": 29825412209049600000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 29825412209049600000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_006000.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_006000.json deleted file mode 100644 index 750d71a2fbb26f1b0337e40ec312fc7126058c22..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_006000.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 6000, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 0.9554082096291466, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 62, - "pos": 91536769, - "epoch": 1, - "pq_idx": 62, - "rg_idx": 91536769 - }, - "loop_state": { - "min_val_bpb": 0.9554082096291466, - "smooth_train_loss": 2.7398973936011246, - "total_training_time": 13281.031767845154, - "stage_training_flops": 32536813318963200000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 32536813318963200000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_006500.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_006500.json deleted file mode 100644 index 4181e367716100c7da4c15dbbd4d60e4cb0badd1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_006500.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 6500, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 0.9459117802045881, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 68, - "pos": 15828769, - "epoch": 1, - "pq_idx": 68, - "rg_idx": 15828769 - }, - "loop_state": { - "min_val_bpb": 0.9459117802045881, - "smooth_train_loss": 2.49100071822428, - "total_training_time": 14388.360385656357, - "stage_training_flops": 35248214428876800000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 35248214428876800000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_007000.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_007000.json deleted file mode 100644 index f5a4e24c3d269d64c60b08b7c70a45ad86aee902..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_007000.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 7000, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 0.9369572412781727, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 73, - "pos": 40120769, - "epoch": 1, - "pq_idx": 73, - "rg_idx": 40120769 - }, - "loop_state": { - "min_val_bpb": 0.9369572412781727, - "smooth_train_loss": 2.562344862017825, - "total_training_time": 15495.668069839478, - "stage_training_flops": 37959615538790400000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 37959615538790400000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_007500.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_007500.json deleted file mode 100644 index a200bb8ab1ebe33204254141848c19b03afd02e9..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_007500.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 7500, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 0.9319988218955536, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 78, - "pos": 64412769, - "epoch": 1, - "pq_idx": 78, - "rg_idx": 64412769 - }, - "loop_state": { - "min_val_bpb": 0.9319988218955536, - "smooth_train_loss": 2.508675269544847, - "total_training_time": 16607.058482646942, - "stage_training_flops": 40671016648704000000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 40671016648704000000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_008000.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_008000.json deleted file mode 100644 index e70cba98f1b0c5fea4206b190e2f5997c10b55f2..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_008000.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 8000, - "training_complete": false, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 0.9235653525987986, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 83, - "pos": 88704769, - "epoch": 1, - "pq_idx": 83, - "rg_idx": 88704769 - }, - "loop_state": { - "min_val_bpb": 0.9235653525987986, - "smooth_train_loss": 2.6049902701313874, - "total_training_time": 17713.889403820038, - "stage_training_flops": 43382417758617600000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 43382417758617600000 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_008352.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_008352.json deleted file mode 100644 index cf3eb850d0f4f95850f74a232ccf48a0e9b1a9b6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/meta_008352.json +++ /dev/null @@ -1,146 +0,0 @@ -{ - "step": 8352, - "training_complete": true, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "val_bpb": 0.9204610080723857, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "wandb_run_id": "e69c1e59", - "wandb_group": "clean1930s-d24", - "wandb_tags": "think-dataset-clean-1930s,d24,ratio12,ctx4096,sssl,fp8", - "device_type": "", - "fp8": true, - "fp8_recipe": "tensorwise", - "depth": 24, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 8, - "total_batch_size": 1048576, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/data", - "tokenizer_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "pretokenized_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/pretok", - "checkpoint_dir": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment_config": "/workspace/nanochat/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "c850ee9513ff3af301af2b7ff127e8709a4f6d49", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" - }, - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 1048576, - "dataloader_state_dict": { - "file_idx": 87, - "pos": 57806337, - "epoch": 1, - "pq_idx": 87, - "rg_idx": 57806337 - }, - "loop_state": { - "min_val_bpb": 0.9204610080723857, - "smooth_train_loss": 2.4791944170448708, - "total_training_time": 18494.615287542343, - "stage_training_flops": 45291244139996774400, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 45291244139996774400 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_000500.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_000500.pt deleted file mode 100644 index 2e91ee326f2fee738ecf5ed8ffc726a26e7b137a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:92b16b12a4f9314cd8f0a61185b7600090193b4fecbf53bcb88da7f74e7b50e5 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_001000.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_001000.pt deleted file mode 100644 index 55f914344986e62d341b8a8d53e7febe1fff3ce5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:15562268665a2d6b5a90170e03e8181ea480f54b7799cf2670cacf86b1198e9f -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_001500.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_001500.pt deleted file mode 100644 index f85d94a14df0a4920d806020200258c4ebf23cfc..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:162b5c3e14416c53ff1d7fb9b3c2563449256e2d38ebd04b64e2a74e17d64dc8 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_002000.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_002000.pt deleted file mode 100644 index a4aa79057aea50e4c121c19874b393279352328d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0534ef4e636c20b5b98f864f7d62ce1f86a14a358e10a6d79bf774ab65260915 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_002500.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_002500.pt deleted file mode 100644 index 4c96772d075bd351beec917b83f40782a08f6a2c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9b6410dee30c9b2be510a624b4bf1f81368696355c2d0ada2169d440c535fb3c -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_003000.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_003000.pt deleted file mode 100644 index e8793ce0caba89ce61afb32fbc25c10d2b9868f9..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_003000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:01b3b851a749fded7a3fb33c341bc7ea0dc1b8a306d61b92559f584da2cca4ca -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_003500.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_003500.pt deleted file mode 100644 index 090af809dfd2d4a08c980f35864c28edb90148b1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_003500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:89c751864d45e7e3f65cb23aebfbaa7597e4f99ad87e6f8ba24a0cd6754f9d59 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_004000.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_004000.pt deleted file mode 100644 index 971ab562f5494f84d8238ef588b2f321c5feaba6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_004000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:27220fe042f0c5975ea3f6c9d6902012e61b5a3d6c11c550c89cd3e5db77e64d -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_004500.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_004500.pt deleted file mode 100644 index fcab921c5559ddfb94b1273612b9926bca4d2c4f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_004500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c8dc23fe6c42bfbcd92a75bf94a5010ff7bb4643753c9caa7fd3a8726819ee2e -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_005000.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_005000.pt deleted file mode 100644 index 3899cf5951c6fa686f69c7130e68cd97fda3c670..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_005000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:37c3fab4507d073152f8d6b139fdc42013196ea66976ef56b220942420019067 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_005500.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_005500.pt deleted file mode 100644 index 2b1d2da0cf31240607dc2915211486458286c03f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_005500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4ecec13dbc90e09c3700a53d9e3b196370189d8028362ff568f740bb4a60584e -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_006000.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_006000.pt deleted file mode 100644 index 4bc7cf2130eff1a18ad07805be8f5355dbf8d01d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_006000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d88ee20a8a5166d3762b1ff285932fb787c3cd42ea49d49388b4e975a25c0951 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_006500.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_006500.pt deleted file mode 100644 index 71420546c5e20396b283dd10e86c81bae02907e2..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_006500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1bf30fdcab49e323beb665e6865166f217b9595e643323d5919872210ee8d2cd -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_007000.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_007000.pt deleted file mode 100644 index 9643016a6618f0b757cb8e458b1a583c99236781..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_007000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0bc91529095a9c9c08ca872f21f3a62b839943224d5bafe9cef7a2107bb0eabe -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_007500.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_007500.pt deleted file mode 100644 index 925ada9093b0a00aecd03ec9e1ce249fe33b0188..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_007500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5953b5db1923843797b98099c9467508e49b285ac2114e6d76ac975d09ada53e -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_008000.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_008000.pt deleted file mode 100644 index 1f4288273a8b899c953514b4a2457b25e3c67161..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_008000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:464aaa00a35555033dc1c0aac2c3d6cb53c4d014410bd013d578ade51fa94b2d -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_008352.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_008352.pt deleted file mode 100644 index 86e83ca5a7bec03e544e11de5e7efa9304a59d07..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/model_008352.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b8fd0aae71fd401972289d32db6bade6335a6caa27952563cf07b2ed3295c664 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_000500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 93585850ef82aeb8a7528b78fa67f15ca92a4cb6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0cc7e57f4bdce3c43531e94027050fa18dc4b03bf57d6f0de66d250e7a409697 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_000500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_000500_rank1.pt deleted file mode 100644 index 345ef5e2b8bbe914cfd3b1295e2c97b2c603a5f1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_000500_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:12e1d3a0a05ff161d1dc5385435be5db467498c99d3c7134d3701a6d99749450 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_000500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_000500_rank2.pt deleted file mode 100644 index ed3c9c0d8a42f8f9486f1631086081a01d9d4951..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_000500_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2369bdf4a0cb9e89d484a15a71768e14056f19805b0bef53c531452269ee7cb3 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_000500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_000500_rank3.pt deleted file mode 100644 index 3b39121e861ed3714a08deb65bfb495bcd1c0739..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_000500_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b4fe8d467d288d8b2993e6182a7013d17afdc1a3d2c6adb1cdd8c649a0862b4a -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 62ee60d2df0853975e4ab44c9837cbe7869c61c2..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3bee737f1c639ccf2be2bf830c17e41120a7b29b67f1c7e178d58ec461e109ca -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001000_rank1.pt deleted file mode 100644 index bc7764bb001679482476c791b6464e89d2f3a9fa..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001000_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8bde52fad137f93d94ff54013f708e56860e65feda91ce71fa65dfeb24f4515e -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001000_rank2.pt deleted file mode 100644 index 9e6d008a29511853039c22292115887ff4b10853..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001000_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:33b997753c16e971ed9c1cfb001f8bf509623b2226af4150b0c8d441e2593986 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001000_rank3.pt deleted file mode 100644 index f3d0232b146e78eca5eac5c5b5b1a6cf376c7425..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001000_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:af2e7798de626e2da15e599e34a7e08ef4cc6b2c939295032cbcd728117d7fd0 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index da1075b1c7fe7d912ca2ebd61529e637bf8c5d8e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d06e0cf87bbb825920142041ccb38e96d0f2acb815da1176d662d87f433554c9 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001500_rank1.pt deleted file mode 100644 index 87ffe0b8ef765b2c2ed42bf33640e94dcf4f64fa..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001500_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a7793dbb1f300bec9a87ae421cd2c8207e620970de175e10a78be8effed439e5 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001500_rank2.pt deleted file mode 100644 index 5261b47ea0f1fc4d34bffdf0e5693dfb071c68cd..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001500_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a91adec757765c848c123101fcf6b413ae1388340a78e4383f0a7a69c6367929 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001500_rank3.pt deleted file mode 100644 index c1c4abed767431659574a6aa7459723fe0fa0275..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_001500_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6d47160736218d41e78fad5e9cd8ec389b5764212fbb58ad84c049bbfeb31cab -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index a5fa9d57bb09d5bd906534b0b560669042792584..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:96e53d566f26d04b10bf736d307a07a8679b3b30712a3018cd9d1bd4d45ce1e9 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002000_rank1.pt deleted file mode 100644 index a37dadbd0acebc188d29bd8bc5b3e49d1fa8a6f5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002000_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:254bcbf387df44b6e659c3653b595d3f7169d19bc3023af8a77d3a36edaaf372 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002000_rank2.pt deleted file mode 100644 index 092872c8bd7e2848782c4566a58f54c540e7e661..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002000_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:87692e55c70f4e174a991d5bbb687efe0fdb92941688dc74adc9ce8d42f09a78 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002000_rank3.pt deleted file mode 100644 index 625638ff678e841052ca699b10aea4a1975cd571..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002000_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:49c0200f7daa52dee3bf9d5b9b7115221efb27cf94f5e6926be6660d3a2d54bb -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002500_rank0.pt deleted file mode 100644 index 3b2372c3f765582e6a66a9aecedae94aada346ee..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:dad53c83ca4abe3e38ad9769694aa60460b4ae0905c97b26fc909b715ff99ada -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002500_rank1.pt deleted file mode 100644 index f08ec1d803e3a0a60814a3db5c3691a1b3e37c4e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002500_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5a34bd1a38bff51a893ecfe24c3717777ff9151383373d9f156fdb9a8815e200 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002500_rank2.pt deleted file mode 100644 index 937a1f26b132ff3169281ca08566c5684f6b10b4..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002500_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2509dc5e95c87dacd6236224301527e70ee090f79587abd9f366d5f07ff426c0 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002500_rank3.pt deleted file mode 100644 index 4180cdbf482ab41cc4c8f4f70bf9cab485a86858..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_002500_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7f78629051b19fb515fb186ac8a10ec7b070fdbc9910eb375aaf81de6167255c -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003000_rank0.pt deleted file mode 100644 index 1582bb06fc848af590f5010884ff306d8d5974e0..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2b36b142dbc92eef6a2b597bea10e50c35792b364f10c739a4dd978d301746ed -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003000_rank1.pt deleted file mode 100644 index 70aac7d9366aa870c14d002e0a524bc7a6aee4c4..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003000_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e499354fed45b3e6463010a7dfe68063137496bb7d557489917a571c6c19a8e5 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003000_rank2.pt deleted file mode 100644 index f072fcf4cb05113fc37614519404c539aab627a1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003000_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:97a972a1a5397ce314164b176357dfe9b1887fdf731ebabbd7a4eb4eb0d6c324 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003000_rank3.pt deleted file mode 100644 index 9e4cc9ce5076a9bb46d962cc845cb6b58e295e4a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003000_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b4ddfdee683483b4abb1950ec913ca8d7f98691df0612802a4123bf4e2704682 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003500_rank0.pt deleted file mode 100644 index abdbc2ed4d4588aac2ef0961d1e56151f2939ea6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7dcd5697de103c0baf3c22a79f28c3f0a21dc54395387cf8e2c36cfd494329b2 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003500_rank1.pt deleted file mode 100644 index 7321d5baa4472f631ad3d5c09655dc92e7b796da..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003500_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:182d936c842dc37db98a4cfda952f091eb28d1d8dc914922954070e567d97b8a -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003500_rank2.pt deleted file mode 100644 index 42b97acd955ddf862030f63f0a09fa06b243e1bc..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003500_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6a0b63b6d56e9f4173bc79e2fc2fadac11bbd2833670d344a03758dfb0815076 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003500_rank3.pt deleted file mode 100644 index c1ee92b374067e34949a6c06a92d260c0071ba62..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_003500_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:77e9e8f44a12b839b4c97e9eabb4a982a295d8630c1e55bed13d0283db529fa6 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004000_rank0.pt deleted file mode 100644 index 14676c035437a7f7da43b55f96bc2a95effa6f0f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bdac937380d94a50045d770e74f0b4c97518bd35369f5930251468d7dfafbe45 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004000_rank1.pt deleted file mode 100644 index 3699b2b65438c61af59e0bc788ac675de529c79f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004000_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7f4c32024d9fe488fa2198a027ae372a45ccde6bb0cf93e24f5f966667de086d -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004000_rank2.pt deleted file mode 100644 index 8fbcd1972122c7ad463a9005bc0b8f2c6efb26bb..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004000_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b539c32e86309a0bdc7f719f72d974bfb85cf7ac9a0904c94c373b5d02d278dd -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004000_rank3.pt deleted file mode 100644 index 65996b9f9637b98015d7fc97cc3730196abc0854..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004000_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7d6a65ff686ad411f508374b430ac13197d66931384dbf72fa17f6cfafb04b05 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank0.pt deleted file mode 100644 index cb0dc888b4af24287478003f5021016684d185b2..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:31be7dce8931348c15493689a6743339a77897ad18c7645f2af3b8de6f81f56b -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank1.pt deleted file mode 100644 index 3fff3d3496de6a76c8c1ea6f6a7b4dd4ccece543..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:eecc5243a2ed93ac65bea8de8373ff27e5208b62b6b17d31b14c9a68eb37394b -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank2.pt deleted file mode 100644 index 13dce95a8d1c26b030f7ecaf731d168b9d13e5c1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:29b34641c29769169a1ed6b4afd58e9e307067cada1e9e88c72966e0144f438d -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank3.pt deleted file mode 100644 index 98dbeeb3f080b71feaff938c500faf89b48b90d8..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c13ffbd7e162bf3c9a022393fd3901c55e8db1cf896cb08c0bdee97c2b6dc83c -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank0.pt deleted file mode 100644 index 7b861d769274333974f8e33297c8f1eb37e321e0..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3ff0cc947e18af1a04987dc2f64fa0a29d22418d34efdff24027a1798c7cea94 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank1.pt deleted file mode 100644 index 9b0368f4cae03c7c2ca414b502cd89ab8eb5cd37..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:085b891a2d75abe7ff9687ec43f89c444604eb83bbf0d5f903b21675eb51160f -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank2.pt deleted file mode 100644 index d042f60072e4692b0a0637901293c48e6d3a5166..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7f20218827152f212517747857afab2b7349aa5a043d96c23f296a5907d717ba -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank3.pt deleted file mode 100644 index 307c8b39a345d1aaf4693a2aa3bfa6553df18ed4..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b94746b3fa690d69a9ba3e5cbde7c7da3e1f0aabb50b73c9f3b1326f4fc38743 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank0.pt deleted file mode 100644 index 97b5eb307d67a3ddc9f52bf474524076744f2983..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:225af4d6d4d822abbce34e63d621eac376be5b8acf857ba8997e36582762573a -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank1.pt deleted file mode 100644 index 08c348506761b7a8f50b88e53f0cbeb838aa5249..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bda69aaa71c4ac52b0c4658c58b7d07f8671608af2f71a512e983da688149ad6 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank2.pt deleted file mode 100644 index 5a4e2e942eef55872dbc96281ed4ff32b842b2ad..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c39f5feb695f84f82218defbe6c535b506173b550fd3ae6fe0caa20278817f3f -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank3.pt deleted file mode 100644 index c61757cd5c711e9694cb916d45a9e3c8199cf222..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b806a61d71740cbb0e2b088ea4c42165334ecd9863494c27a343b548b0d84326 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank0.pt deleted file mode 100644 index fd2f5d56db044859c812df5fd2bbfd059d3bd21d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f7e168383bc13bbc56df91c21f3d46e835001ed592f90f849af53993f9c965a3 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank1.pt deleted file mode 100644 index 9d7907599f9c8c5ceaf91e5db771254a97b510f9..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a068ed26f38b48c5c7ef70a1771da690ab6cdd1749d03f902277b5b2a553f94e -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank2.pt deleted file mode 100644 index a7b95224bd425230871a29870c5278c70400b007..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f63a2df3dbbe97e7bbf952c0fc0eb732423d49fe7e994c9b632e1342d9ac8868 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank3.pt deleted file mode 100644 index 972682cada1c58e72a6fcdf9f0212fbf56440dc9..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:60613cd93ddac30dd28676e47ae16588c2ad85aa09b9bd4f63f17a5ad73dfc07 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank0.pt deleted file mode 100644 index 9f93afed1bd0ecb6105660308bc2a5a3d5ccbd01..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7a5fd4763414c513d80406c35ecdf9f26ec5d1d9d27a676e39dfd01c8e6992b7 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank1.pt deleted file mode 100644 index 4bef645d0b66dfee1f1d7fd0d2b2f4e9582d351b..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7b21033cf7c8929fe01be78ca45b157751d65c7dff5cb57940226319a9d76d9c -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank2.pt deleted file mode 100644 index 79f3c8fe08e743ead381bcfad755104938ba57a1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:99ac47912efb2feda2492d9f772370071b090843ee13e55a08e1e31d7068926a -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank3.pt deleted file mode 100644 index 9d412edbc4b071b6222f9bb84677d8c05361eb31..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:10a5f118f4f3537eaf2230fbeef6fe6e9298958787e807116fadf09f5fc0e04e -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank0.pt deleted file mode 100644 index 8ae39270d361a274d0537b6f5cea28b86b715ba7..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d27d602d30b1a4d7aa9605fa175739fb161b32e02909ec30c231e3fde3be8264 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank1.pt deleted file mode 100644 index be238b1077c4b9f3007287b37c203d663d72a251..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e638779e6afd341c5453a5de23e7c834f305837bd70cc5a0fc12b277cd7ee9fc -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank2.pt deleted file mode 100644 index 7dd7b85740cb95e60a4008b72b3aa3077437a4ce..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a1afe22caa4f2819cfc397846b74b8c26272f6266a16746d31f3a87bac3a5c37 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank3.pt deleted file mode 100644 index 06a116d9c933a6dc0e2dbe3234b5816cd49a810e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:acbcb96c068e6d855d7619f22c370140f31a894c4d23f105c7522214d8a51ae4 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank0.pt deleted file mode 100644 index b1537d401d561c880d1faa625adc81be8aafad24..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7c8beda301552c587b21e718b3f0eb383006dbb232a26d5951b9132153346c4d -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank1.pt deleted file mode 100644 index 7790f7ac35002bbe773082b795089f511cc02e95..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7db4c7619d2c479f1b6a41805b4fc74865c7f4563cf71d8105af4ecc6f9ffd27 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank2.pt deleted file mode 100644 index 42a78a54b5c4468dc66fbc6b0b8b3877c3bdfaee..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:cdc736fbeef52ca1bfe7e975b93ad550bfc35129ba2e25eddad9ecf59d3fc3bf -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank3.pt deleted file mode 100644 index d05e178495cd94a09a04506f0b343e69b808a9ba..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:30523b0f3a84b56579b1843151587d357fbae481de2e0885f92b484ed6356e2e -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank0.pt deleted file mode 100644 index 2858604c969f00c2fddb45e4209e5c0802aa9585..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:11ff33d778dc9ba30dc3fda9174bfe1b73fd251370a901f1e396ab7de520028e -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank1.pt deleted file mode 100644 index 20b10659b1b4b4e09e6187fb9cac0b7e7b1d852c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fa40a10a4058f06d01d3393321e236973fa78f2d137d1e02d7ca4c119b7305e6 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank2.pt deleted file mode 100644 index d842272f70f6f28b6c98f6a116a587c40673b354..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:22d887c0fc65e71df1f3d74b5ea3656ca860658d8249aa8cac96421e89a7a835 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank3.pt deleted file mode 100644 index 79f86732149bf73e5dfb1bb3886e49c01c8ea6e8..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fc131b1f33fa81372bcc1ecb759f2b7ddf2acf55652b07e1a14b26b96a909e73 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank0.pt deleted file mode 100644 index 5670db0bbfcac33c0e5aa71e025e54e9d33a34e7..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:ef67bbc1c6e6f73cf3ccfe50d3df3e916b2cf913a2ef1fb6a14123e784e4bc13 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank1.pt deleted file mode 100644 index 230997d0dc5116c8c5be6dda4c3b08009e4236d7..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank1.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:839d78c2d9e329f91167b3ca2a706ba931d4fc946101826a6bbac664b3c366b1 -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank2.pt deleted file mode 100644 index 5a6ae573964c5e34aaad1fdc5b22e1df2004bd78..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank2.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7987c59cd7ef08458fcdf58f4e737501e3adf2e5a32cd0e44015ba70ee68d29b -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank3.pt deleted file mode 100644 index 7ec76ac70790641f6de6c4aa61c6124217a02a89..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank3.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4b5c08a5e6fd49981ac6887b43d4d835742ef20e8eefa060670635dedc0c69de -size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json deleted file mode 100644 index 83f912760475957e23448e8669d54495f0cc358d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json +++ /dev/null @@ -1,63 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 180, - "download_workers": 4 - }, - "tokenizer": { - "mode": "reuse", - "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 24, - "scaling_params": 729810624, - "target_param_data_ratio": 12, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 1048576, - "fp8": true, - "fp8_recipe": "tensorwise", - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano", - "keep_local_checkpoints": 1 - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "group": "clean1930s-d24", - "tags": [ - "think-dataset-clean-1930s", - "d24", - "ratio12", - "ctx4096", - "sssl", - "fp8" - ] - }, - "config_fingerprint": "166e7e695c1b536a", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/core.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/core.json deleted file mode 100644 index 9dd677469d98b2c43c74fc93887dafa421454807..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 8352)", - "step": 8352, - "bpb": {}, - "core_metric": 0.13542864491271994, - "core_results": { - "hellaswag_zeroshot": 0.32364070415496826, - "jeopardy": 0.0037789323832839727, - "bigbench_qa_wikidata": 0.2516116201877594, - "arc_easy": 0.41750839352607727, - "arc_challenge": 0.239761084318161, - "copa": 0.6100000143051147, - "commonsense_qa": 0.3071253001689911, - "piqa": 0.5805222988128662, - "openbook_qa": 0.2580000162124634, - "lambada_openai": 0.35338637232780457, - "hellaswag": 0.32911768555641174, - "winograd": 0.6703296899795532, - "winogrande": 0.49408048391342163, - "bigbench_dyck_languages": 0.12700000405311584, - "agi_eval_lsat_ar": 0.260869562625885, - "bigbench_cs_algorithms": 0.38333332538604736, - "bigbench_operators": 0.11428572237491608, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.114096499979496, - "coqa": 0.16610296070575714, - "boolq": 0.5969418883323669, - "bigbench_language_identification": 0.25699999928474426 - }, - "centered_results": { - "hellaswag_zeroshot": 0.09818760553995769, - "jeopardy": 0.0037789323832839727, - "bigbench_qa_wikidata": 0.2516116201877594, - "arc_easy": 0.22334452470143637, - "arc_challenge": -0.013651887575785318, - "copa": 0.2200000286102295, - "commonsense_qa": 0.13390662521123883, - "piqa": 0.16104459762573242, - "openbook_qa": 0.010666688283284506, - "lambada_openai": 0.35338637232780457, - "hellaswag": 0.105490247408549, - "winograd": 0.34065937995910645, - "winogrande": -0.011839032173156738, - "bigbench_dyck_languages": 0.12700000405311584, - "agi_eval_lsat_ar": 0.07608695328235625, - "bigbench_cs_algorithms": 0.38333332538604736, - "bigbench_operators": 0.11428572237491608, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.114096499979496, - "coqa": 0.16610296070575714, - "boolq": -0.060679241230613294, - "bigbench_language_identification": 0.1826182610393226 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/samples.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/samples.json deleted file mode 100644 index 2b91acc601be03a4af30318151f8cf79249cf54d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 8352)", - "step": 8352, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the city of Paris, which is the capital of France. The capital of France" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is Au, and the symbol of silver is Ag. The symbol of gold is Au" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If yesterday was Friday, then tomorrow will be Saturday. If yesterday was" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is cold.\n\nThe opposite of cold is hot.\n\nThe opposite of hot is cold.\n\n" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: Mercury, Venus, Earth, Mars, Jupiter, Saturn, Uranus," - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a dark brown, with a tinge of red. It is a very pretty color" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is 13, and the equation is 13x + 3 = 13" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nTHIS essay was contributed to the \"Psychological Bulletin,\" of the University of Chicago, during the past summer; the temporary arrangements for its printing prevented its meeting the demand for it. Fair play in the matter of results in the revision of Shakespeare's plays has not been given in this field in this company, Dunning, Uzziel, and others engaging to contribute whatever they have to say on the revision and construction of even a very few plays, and the magazine enterprise to furnish more plays of a classic nature, although comprising many dramas of merit and of the utmost possible variety, is by no means of sufficient scope and interest", - "<|bos|>\n\nARTON LIBRARY OF THE DEZA SUPERINTENDENTS AND BOOKKEEPERS\n\nThe following Outline of Business\n\nthe State or Territory in which\n\nFoss work is to be carried on.\n\nPLAN OF BUSINESS Department 1. Traveling\n\n2. Records, coupons, checks, mercantile accounts, vouchers, &c.\n\n3. General office\n\n4. Subsidiary agencies. Persons dealing with department men.\n\nPlanwork.\n\nPlan.\n\n1. Plans for the method of review of machinery, etc. Uncertainty.\n\n2. Plans for securing uniformity of practice over a large field", - "<|bos|>!\n\nIt's the coss is mighty wonderous weak Thefe minutes yhere hain marry teef The tryfector\n\nOf my fandd in Life who ought to hast The pist o' a President\n\nApril 10. J,\"\n\nAmong the Arms in cison\n\nThe Arms of the President *The Arms of Washington\n\nOn October 5, (?) in another and more noticeable storm-swept plagiarism, the President bore the baptismal emblem of the United States; on the way to Washington, January 1, 1799, occurred the cancel that has made possible the interpolations of the following", - "<|bos|> ministered unto Him with great joy.\"\n\nXX. 1. CHIRIT OF THE BLESSED.\n\nThe full contrast to this joy is the hunger that followed when He left His throne in heaven, ascended, and took the place of the \"Name\" lifted up on Calvary.\n\n2. The Fulfilment of the Tabernacle: In our imagination let 50 Enter: first, the Church universal: prayer, glorification, rejoicing, nothing but an eternal army and the eternal joy which forms one side of this-and well! for consequently this is that infant Church which a voice descended, and the Redeemer Sunday broke through the wall that", - "<|bos|>eed\n\nNew York paid\n\nEach separate $1.25 and i piece by akely ried in Watts averaged 75 cents.\n\nRetail price 50 cents-No longer find it worth while to order second-hand books. Already considerable order great publishing interests are sending agents to England or Scotland to our mmission depot expressly to collect the $1.25 per set, on which income is made over to us. The influens can gain fifteen per cent, and this can be taken care of at once as it is here almost never debited. Its great advantages that has the most of A. L. Burlingame's", - "<|bos|>PREFACE\n\nWHEN PERSHAD was about to issue and was being attacked by the fanatical cuttlefish of Northern India, some weeks before at least the Calcutta of 1862 stood face to face with a panic-crazed gale, the doctors began one day on their platform to advise the world how they were going to appease the spirit of the Fish-god. The Elders and the Lady-Seconds criticized emphatically the pros and cons, and the whole city stared.\n\nOne man said: \"I am Oriental; the people of this English country do not believe in all this agnostic nonsense, and the skies when they are", - "<|bos|>iman have published,\n\nTo be completed in about 100 Monthly Parts, each illustrated by a Steel Engraving, a Work consisting of a Library of Original 180 Engravings, 3 vols. 8vo. \u00a35. 5s. strongly bound in cloth, as a Work for convenient\n\nSale by STEEL Engraved by W. L. STEENERSON, each neatly illustrated by a Steel Engraving; the Plates are in a very high state of\n\nPerfection, and warranted to give a correct Image of each Novel.\n\nThe Work is published in Monthly Parts to contain Historical and Biographical Sketches of the principal", - "<|bos|>MYSTER LIBRARY.\n\n3 2044 009 603 166 fidh\n\nQUE\n\n7808, W" - ] -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/val_bpb.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/val_bpb.json deleted file mode 100644 index f488ffe394f042d6c23f4e7edd1b3cf593fc9321..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 8352)", - "step": 8352, - "bpb": { - "val": 0.8435203902195944 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/run.json deleted file mode 100644 index 255cd14f4914c9569b6ce62fb3ed77dfeb8b012f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "166e7e695c1b536a", - "wandb_run_id": "e69c1e59", - "created_at": 1784557543 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1/config.json deleted file mode 100644 index fc2b5b91b9afdcdf7b0dfcfb9e7621733d063c85..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1/config.json +++ /dev/null @@ -1,36 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "nanochat-default-v1", - "data": { - "recipe": "nanochat-default", - "mmlu_epochs": 3, - "gsm8k_epochs": 4 - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "eval_every": -1, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": 200 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": false, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "clean1930s-d24-r12", - "tags": [ - "sft", - "smoltalk", - "mmlu3", - "gsm8k4" - ] - }, - "config_fingerprint": "103d7c3526aa35c0", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1" -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1/run.json deleted file mode 100644 index c04b45d5fdfbfc4203f0f441cf98173107e575e0..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1", - "stage": "sft", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 8352, - "branch_parent_step": null, - "config_fingerprint": "103d7c3526aa35c0", - "wandb_run_id": null, - "created_at": 1786512279 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/meta_000007.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/meta_000007.json deleted file mode 100644 index c376b67eb26d26eca43d12d72a50e81853709ea1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/meta_000007.json +++ /dev/null @@ -1,103 +0,0 @@ -{ - "step": 7, - "training_complete": true, - "val_bpb": 0.7205382880492125, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic", - "wandb_run_id": "0837926f", - "wandb_group": "think-d12", - "wandb_tags": "sft,pre1930,ratio20", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "base_step": 8352, - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "resume_from_step": null, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/config.json", - "parent_cumulative_flops": 4.5291244139996774e+19, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "60aa7005265901858554bb7ca98cd788de721e4a", - "load_optimizer": 1, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.0, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": -1, - "eval_tokens": 20971520, - "chatcore_every": -1, - "chatcore_max_cat": -1, - "chatcore_max_sample": 24, - "save_every": -1, - "recipe": "pre1930", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-authentic", - "data": { - "recipe": "pre1930", - "pre1930_epochs": 5 - }, - "training": { - "num_iterations": -1, - "device_batch_size": 8, - "eval_every": -1, - "chatcore_every": -1, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "pre1930", - "ratio20" - ] - }, - "config_fingerprint": "d3378357cef17359", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "d3378357cef17359" - }, - "loop_state": { - "step": 7, - "total_training_time": 0.0, - "min_val_bpb": 0.7205382880492125, - "smooth_train_loss": 1.187201474390888, - "mfu": 35.33209107765201, - "tok_per_sec": 21315, - "stage_training_flops": 3.79596155387904e+16, - "inherited_parent_flops": 4.5291244139996774e+19, - "cumulative_pipeline_training_flops": 4.5329203755535565e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/model_000007.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/model_000007.pt deleted file mode 100644 index 93188a16a277a2a7ae8c9dfb8219b9f85520cb74..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/model_000007.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3c16eb054f43f54f05624b42a4ec3fbcab666b80d99c3c87ac6ee8b34261a131 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/optim_000007_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/optim_000007_rank0.pt deleted file mode 100644 index cc9feffc054f9812e3c1890c16ee483d079daf05..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/optim_000007_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c53d48431c7459eae0e562672d69b2dc15a15f91e1243ce120d54ab6b64ba458 -size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/config.json deleted file mode 100644 index 1b97707ee9231149762d0f88f4d8ba80b96831b8..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/config.json +++ /dev/null @@ -1,32 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-authentic", - "data": { - "recipe": "pre1930", - "pre1930_epochs": 5 - }, - "training": { - "num_iterations": -1, - "device_batch_size": 8, - "eval_every": -1, - "chatcore_every": -1, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "pre1930", - "ratio20" - ] - }, - "config_fingerprint": "d3378357cef17359", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic" -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/run.json deleted file mode 100644 index 8ba5b17470a8d0c9e6a7acae2e34be4f07c4e9b1..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic", - "stage": "sft", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": null, - "config_fingerprint": "d3378357cef17359", - "wandb_run_id": "0837926f", - "created_at": 1784577842 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/eval_metrics.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/eval_metrics.json deleted file mode 100644 index f4856feb918ed0240238d249b6b784926bf0b743..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/eval_metrics.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0", - "recipe": "curriculum", - "step": 52, - "val_bpb": 0.8869868096604953, - "min_val_bpb": 0.8464286315951827, - "per_route_bpb": {}, - "per_domain_bpb": {}, - "curriculum_summary": null, - "chatcore": { - "chatcore_metric": -0.006895505470088888, - "chatcore_cat": -0.011492509116814813, - "suite": { - "tasks": [ - "ARC-Easy", - "ARC-Challenge", - "MMLU", - "GSM8K", - "SpellingBee" - ], - "max_generative_problems": 32, - "generative_answer_format": "End your response with #### followed by the final numeric answer (for example: #### 42)." - }, - "ARC-Easy": 0.24494949494949494, - "ARC-Challenge": 0.2354948805460751, - "MMLU": 0.24369747899159663, - "GSM8K": 0.0, - "SpellingBee": 0.0 - } -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/meta_000052.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/meta_000052.json deleted file mode 100644 index 566d3fd5fd3bd34383abba37c5aa0f6776d725c3..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/meta_000052.json +++ /dev/null @@ -1,161 +0,0 @@ -{ - "step": 52, - "training_complete": true, - "val_bpb": 0.8869868096604953, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0", - "wandb_run_id": "fb8277f3", - "wandb_group": "think-d12", - "wandb_tags": "sft,curriculum,c0,baseline", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "base_step": 8352, - "checkpoint_dir": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints", - "tokenizer_dir": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "resume_from_step": null, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0", - "experiment_config": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/config.json", - "parent_cumulative_flops": 4.5291244139996774e+19, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "", - "load_optimizer": 0, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.03, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": 200, - "eval_tokens": 20971520, - "chatcore_every": -1, - "chatcore_max_cat": -1, - "chatcore_max_sample": 24, - "save_every": -1, - "recipe": "curriculum", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c0", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C0", - "mode": "flat", - "epochs": 3, - "threshold_default": 90, - "routes": { - "knowledge_qa": { - "count": 18000 - }, - "multiturn_qa": { - "count": 18000 - }, - "reasoning_qa": { - "count": 12000 - }, - "narrative_grounded": { - "count": 12000 - }, - "opinion_qa": { - "count": 10000 - }, - "composition_qa": { - "count": 10000 - }, - "how_to_qa": { - "count": 7000 - }, - "verse_qa": { - "count": 6000 - }, - "narrative_fiction": { - "count": 6000 - }, - "stem_reasoning": { - "count": 4000 - } - }, - "calibration_qa": { - "count": null - }, - "authentic": { - "count": 12257 - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 200, - "chatcore_every": -1, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c0", - "baseline" - ] - }, - "config_fingerprint": "4ce60ef1ccb8a8c8", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "4ce60ef1ccb8a8c8" - }, - "loop_state": { - "step": 52, - "total_training_time": 2071.3676829338074, - "min_val_bpb": 0.8464286315951827, - "smooth_train_loss": 1.608637469780549, - "mfu": 35.219659061221186, - "tok_per_sec": 21247, - "stage_training_flops": 2.819857154310144e+17, - "inherited_parent_flops": 4.5291244139996774e+19, - "cumulative_pipeline_training_flops": 4.557322985542779e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/model_000052.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/model_000052.pt deleted file mode 100644 index 19d84e58368f8e1ef9b337f84c76056060ec3d5c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/model_000052.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:70c91228c289d6b4d6928d866710be6bda73f21a0e8c1f2a772cd0cd593e2806 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/optim_000052_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/optim_000052_rank0.pt deleted file mode 100644 index 4dd39a75dd77093bc084924d8e43e762567e83cb..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/optim_000052_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6e93a9ac2c446008d78c0697dcc6b159bdb9504247ae98a271535459cd527493 -size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/config.json deleted file mode 100644 index c3133e425f32aeabd331a214cf024a2d2e7540e4..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/config.json +++ /dev/null @@ -1,78 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c0", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C0", - "mode": "flat", - "epochs": 3, - "threshold_default": 90, - "routes": { - "knowledge_qa": { - "count": 18000 - }, - "multiturn_qa": { - "count": 18000 - }, - "reasoning_qa": { - "count": 12000 - }, - "narrative_grounded": { - "count": 12000 - }, - "opinion_qa": { - "count": 10000 - }, - "composition_qa": { - "count": 10000 - }, - "how_to_qa": { - "count": 7000 - }, - "verse_qa": { - "count": 6000 - }, - "narrative_fiction": { - "count": 6000 - }, - "stem_reasoning": { - "count": 4000 - } - }, - "calibration_qa": { - "count": null - }, - "authentic": { - "count": 12257 - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 200, - "chatcore_every": -1, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c0", - "baseline" - ] - }, - "config_fingerprint": "4ce60ef1ccb8a8c8", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0" -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/evals/chatcore.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/evals/chatcore.json deleted file mode 100644 index 4c56dd37bf29d9e77abae24b1e0c752bfd313a88..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/evals/chatcore.json +++ /dev/null @@ -1,27 +0,0 @@ -{ - "stage": "sft", - "step": 52, - "total_training_flops": 4.557322985542779e+19, - "stage_training_flops": 2.819857154310144e+17, - "inherited_parent_flops": 4.5291244139996774e+19, - "cumulative_pipeline_training_flops": 4.557322985542779e+19, - "results": { - "ARC-Easy": 0.24494949494949494, - "ARC-Challenge": 0.2354948805460751, - "MMLU": 0.24369747899159663, - "GSM8K": 0.0, - "SpellingBee": 0.0 - }, - "chatcore_metric": -0.006895505470088888, - "chatcore_suite": { - "tasks": [ - "ARC-Easy", - "ARC-Challenge", - "MMLU", - "GSM8K", - "SpellingBee" - ], - "max_generative_problems": 32, - "generative_answer_format": "End your response with #### followed by the final numeric answer (for example: #### 42)." - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/run.json deleted file mode 100644 index d3d9fd3be15bed080d5915db0d0c022f1bf0f058..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/run.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0", - "wandb_run_id": "fb8277f3", - "created_at": 1786488813, - "recovered_from_checkpoint_step": 52, - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 8352 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/meta_000003.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/meta_000003.json deleted file mode 100644 index 45cd36180ac79aef95b828b412607b6a268bf1e4..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/meta_000003.json +++ /dev/null @@ -1,181 +0,0 @@ -{ - "step": 3, - "training_complete": true, - "val_bpb": 0.8281267785987712, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust", - "wandb_run_id": "d8812a13", - "wandb_group": "think-d12", - "wandb_tags": "sft,curriculum,c1,minimalist,lima,robustness,noise", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "base_step": 8352, - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "resume_from_step": null, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/config.json", - "parent_cumulative_flops": 4.5291244139996774e+19, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "e315bfb5d50d37456aea0c20a8e8a9109c12f9c7", - "load_optimizer": 0, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.03, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": 100, - "eval_tokens": 20971520, - "chatcore_every": 100000, - "chatcore_max_cat": -1, - "chatcore_max_sample": 24, - "save_every": -1, - "recipe": "curriculum", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c1-robust", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C1R", - "mode": "flat", - "epochs": 4, - "threshold_default": 97, - "routes": { - "knowledge_qa": { - "count": 600 - }, - "multiturn_qa": { - "count": 500 - }, - "reasoning_qa": { - "count": 400 - }, - "narrative_grounded": { - "count": 400 - }, - "opinion_qa": { - "count": 400 - }, - "composition_qa": { - "count": 400 - }, - "how_to_qa": { - "count": 300 - }, - "stem_reasoning": { - "count": 300 - }, - "verse_qa": { - "count": 300 - }, - "narrative_fiction": { - "count": 200 - } - }, - "calibration_qa": { - "count": 300 - }, - "authentic": { - "count": 1000 - }, - "noise": { - "rate": 0.3 - }, - "robustness": { - "epochs": 2, - "routes": { - "conversation_qa": { - "count": 300 - }, - "unparseable_qa": { - "count": 200 - }, - "era_qa": { - "count": 100 - } - } - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 100, - "chatcore_every": 100000, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c1", - "minimalist", - "lima", - "robustness", - "noise" - ] - }, - "config_fingerprint": "6dd68331f59b8eb4", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6dd68331f59b8eb4" - }, - "loop_state": { - "step": 3, - "total_training_time": 0.0, - "min_val_bpb": 0.8281267785987712, - "smooth_train_loss": 0.7139463255405425, - "mfu": 35.07294255982036, - "tok_per_sec": 21159, - "stage_training_flops": 1.62684066594816e+16, - "inherited_parent_flops": 4.5291244139996774e+19, - "cumulative_pipeline_training_flops": 4.530751254665626e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/model_000003.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/model_000003.pt deleted file mode 100644 index 26b07ebde9cfdcd2f54a8ad87c542dc298aeca2c..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/model_000003.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a10e1430b0ab2b416762c891acf5c8d0fc987e7272a4086266bd096fd80d6dec -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/optim_000003_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/optim_000003_rank0.pt deleted file mode 100644 index a9850986237c9f0951f2877d96d8901b941c3642..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/optim_000003_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:776dabeeabf2167abc5be17a01b6ab81a4aab9b91fa5eb69a15ad1136522a77b -size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/config.json deleted file mode 100644 index 02a1d75ec94de4f8df93ed5fd9119926db23529d..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/config.json +++ /dev/null @@ -1,98 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c1-robust", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C1R", - "mode": "flat", - "epochs": 4, - "threshold_default": 97, - "routes": { - "knowledge_qa": { - "count": 600 - }, - "multiturn_qa": { - "count": 500 - }, - "reasoning_qa": { - "count": 400 - }, - "narrative_grounded": { - "count": 400 - }, - "opinion_qa": { - "count": 400 - }, - "composition_qa": { - "count": 400 - }, - "how_to_qa": { - "count": 300 - }, - "stem_reasoning": { - "count": 300 - }, - "verse_qa": { - "count": 300 - }, - "narrative_fiction": { - "count": 200 - } - }, - "calibration_qa": { - "count": 300 - }, - "authentic": { - "count": 1000 - }, - "noise": { - "rate": 0.3 - }, - "robustness": { - "epochs": 2, - "routes": { - "conversation_qa": { - "count": 300 - }, - "unparseable_qa": { - "count": 200 - }, - "era_qa": { - "count": 100 - } - } - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 100, - "chatcore_every": 100000, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c1", - "minimalist", - "lima", - "robustness", - "noise" - ] - }, - "config_fingerprint": "6dd68331f59b8eb4", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust" -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/run.json deleted file mode 100644 index c4ec6de669fa94556cdde3f1adc1a8c5488fada4..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust", - "stage": "sft", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": null, - "branch_parent_step": null, - "config_fingerprint": "6dd68331f59b8eb4", - "wandb_run_id": "d8812a13", - "created_at": 1786564717 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/eval_metrics.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/eval_metrics.json deleted file mode 100644 index 0f7b7f263f49b9481015744d74dbeabb5c25f9bf..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/eval_metrics.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1", - "recipe": "curriculum", - "step": 3, - "val_bpb": 0.827509209527302, - "min_val_bpb": 0.827509209527302, - "per_route_bpb": {}, - "per_domain_bpb": {}, - "curriculum_summary": null, - "chatcore": { - "chatcore_metric": -0.009923963742371995, - "chatcore_cat": -0.016539939570619992, - "suite": { - "tasks": [ - "ARC-Easy", - "ARC-Challenge", - "MMLU", - "GSM8K", - "SpellingBee" - ], - "max_generative_problems": 32, - "generative_answer_format": "End your response with #### followed by the final numeric answer (for example: #### 42)." - }, - "ARC-Easy": 0.25252525252525254, - "ARC-Challenge": 0.22866894197952217, - "MMLU": 0.2315909414613303, - "GSM8K": 0.0, - "SpellingBee": 0.0 - } -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/meta_000003.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/meta_000003.json deleted file mode 100644 index 1bac04363882d6eeaa798629684ca35006f8aa13..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/meta_000003.json +++ /dev/null @@ -1,162 +0,0 @@ -{ - "step": 3, - "training_complete": true, - "val_bpb": 0.827509209527302, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1", - "wandb_run_id": "929f01d4", - "wandb_group": "think-d12", - "wandb_tags": "sft,curriculum,c1,minimalist,lima", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "base_step": 8352, - "checkpoint_dir": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints", - "tokenizer_dir": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "resume_from_step": null, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1", - "experiment_config": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/config.json", - "parent_cumulative_flops": 4.5291244139996774e+19, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "", - "load_optimizer": 0, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.03, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": 100, - "eval_tokens": 20971520, - "chatcore_every": 100000, - "chatcore_max_cat": -1, - "chatcore_max_sample": 24, - "save_every": -1, - "recipe": "curriculum", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c1", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C1", - "mode": "flat", - "epochs": 4, - "threshold_default": 97, - "routes": { - "knowledge_qa": { - "count": 600 - }, - "multiturn_qa": { - "count": 500 - }, - "reasoning_qa": { - "count": 400 - }, - "narrative_grounded": { - "count": 400 - }, - "opinion_qa": { - "count": 400 - }, - "composition_qa": { - "count": 400 - }, - "how_to_qa": { - "count": 300 - }, - "stem_reasoning": { - "count": 300 - }, - "verse_qa": { - "count": 300 - }, - "narrative_fiction": { - "count": 200 - } - }, - "calibration_qa": { - "count": 300 - }, - "authentic": { - "count": 1000 - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 100, - "chatcore_every": 100000, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c1", - "minimalist", - "lima" - ] - }, - "config_fingerprint": "da4f59f54e9dac3b", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "da4f59f54e9dac3b" - }, - "loop_state": { - "step": 3, - "total_training_time": 0.0, - "min_val_bpb": 0.827509209527302, - "smooth_train_loss": 0.705148291826248, - "mfu": 35.12240192928327, - "tok_per_sec": 21189, - "stage_training_flops": 1.62684066594816e+16, - "inherited_parent_flops": 4.5291244139996774e+19, - "cumulative_pipeline_training_flops": 4.530751254665626e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/model_000003.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/model_000003.pt deleted file mode 100644 index 8fee18401fffc1fed7f35b4efbee8fd9710cfc12..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/model_000003.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d615b5c5f303e2177711c7c151a28953692d76060cac82f38803443c9a62c62a -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/optim_000003_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/optim_000003_rank0.pt deleted file mode 100644 index aec65a9096aca26b77b5648dffe4a17e308c46f3..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/optim_000003_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8c69c388a2cb60df3b4c0e0c12951a981f6de5ae19f5b1c1742d0080628a2491 -size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/config.json deleted file mode 100644 index 28d06f8c4549f6378abda73e6d02529dfde31da6..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/config.json +++ /dev/null @@ -1,79 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c1", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C1", - "mode": "flat", - "epochs": 4, - "threshold_default": 97, - "routes": { - "knowledge_qa": { - "count": 600 - }, - "multiturn_qa": { - "count": 500 - }, - "reasoning_qa": { - "count": 400 - }, - "narrative_grounded": { - "count": 400 - }, - "opinion_qa": { - "count": 400 - }, - "composition_qa": { - "count": 400 - }, - "how_to_qa": { - "count": 300 - }, - "stem_reasoning": { - "count": 300 - }, - "verse_qa": { - "count": 300 - }, - "narrative_fiction": { - "count": 200 - } - }, - "calibration_qa": { - "count": 300 - }, - "authentic": { - "count": 1000 - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 100, - "chatcore_every": 100000, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c1", - "minimalist", - "lima" - ] - }, - "config_fingerprint": "da4f59f54e9dac3b", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1" -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/evals/chatcore.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/evals/chatcore.json deleted file mode 100644 index ebc8c1a7e19b73e6e84f6d7ac3f3ee2aeeb85a22..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/evals/chatcore.json +++ /dev/null @@ -1,27 +0,0 @@ -{ - "stage": "sft", - "step": 3, - "total_training_flops": 4.530751254665626e+19, - "stage_training_flops": 1.62684066594816e+16, - "inherited_parent_flops": 4.5291244139996774e+19, - "cumulative_pipeline_training_flops": 4.530751254665626e+19, - "results": { - "ARC-Easy": 0.25252525252525254, - "ARC-Challenge": 0.22866894197952217, - "MMLU": 0.2315909414613303, - "GSM8K": 0.0, - "SpellingBee": 0.0 - }, - "chatcore_metric": -0.009923963742371995, - "chatcore_suite": { - "tasks": [ - "ARC-Easy", - "ARC-Challenge", - "MMLU", - "GSM8K", - "SpellingBee" - ], - "max_generative_problems": 32, - "generative_answer_format": "End your response with #### followed by the final numeric answer (for example: #### 42)." - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/run.json deleted file mode 100644 index dcf9681ef472a6272bfd8c19accbd3955f0c3dd5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/run.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1", - "wandb_run_id": "929f01d4", - "created_at": 1786489421, - "recovered_from_checkpoint_step": 3, - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 8352 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/meta_000040.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/meta_000040.json deleted file mode 100644 index 23c0d9ea09b5fc41699996348d9008c104f20ad0..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/meta_000040.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 40, - "training_complete": true, - "val_bpb": 0.8800241308853763, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2", - "wandb_run_id": "69545234", - "wandb_group": "think-d12", - "wandb_tags": "sft,curriculum,c2,reasoning-forward", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "base_step": 8352, - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "resume_from_step": null, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/config.json", - "parent_cumulative_flops": 4.5291244139996774e+19, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "59ab2594ad6fe85e4294aada37a6f0a0284eebda", - "load_optimizer": 0, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.03, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": 200, - "eval_tokens": 20971520, - "chatcore_every": 100000, - "chatcore_max_cat": -1, - "chatcore_max_sample": 32, - "save_every": -1, - "recipe": "curriculum", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c2", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C2", - "mode": "flat", - "epochs": 3, - "threshold_default": 90, - "routes": { - "reasoning_qa": { - "count": 16000 - }, - "how_to_qa": { - "count": null - }, - "stem_reasoning": { - "count": null, - "threshold": 80 - }, - "knowledge_qa": { - "count": 10000 - }, - "multiturn_qa": { - "count": 10000 - }, - "narrative_grounded": { - "count": 6000 - }, - "opinion_qa": { - "count": 6000 - }, - "composition_qa": { - "count": 6000 - }, - "verse_qa": { - "count": 3000 - }, - "narrative_fiction": { - "count": 3000 - } - }, - "calibration_qa": { - "count": null - }, - "authentic": { - "count": 12257 - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 200, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c2", - "reasoning-forward" - ] - }, - "config_fingerprint": "74c02b2162b87713", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "74c02b2162b87713" - }, - "loop_state": { - "step": 40, - "total_training_time": 1483.5767045021057, - "min_val_bpb": 0.8465480228052693, - "smooth_train_loss": 1.6970454443686784, - "mfu": 35.06659423129714, - "tok_per_sec": 21155, - "stage_training_flops": 2.16912088793088e+17, - "inherited_parent_flops": 4.5291244139996774e+19, - "cumulative_pipeline_training_flops": 4.550815622878986e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/model_000040.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/model_000040.pt deleted file mode 100644 index 2118defd9129f117ad18b1de1e7f1dc5b786da1a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/model_000040.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f9ef0df02226735f57dc26525d46aa2be920b67bdcc6244f7a1475f10b6b3e00 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/optim_000040_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/optim_000040_rank0.pt deleted file mode 100644 index ba948db1373686190827e009d1fd75b0f14fa709..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/optim_000040_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c0c41f8550f90563488e72cb3afd0244964901c668f2e4b552979d23d0a21daa -size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/config.json deleted file mode 100644 index 3a022d15ffd4cfa504d7f3ca9fd2567ae12f62cc..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/config.json +++ /dev/null @@ -1,80 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c2", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C2", - "mode": "flat", - "epochs": 3, - "threshold_default": 90, - "routes": { - "reasoning_qa": { - "count": 16000 - }, - "how_to_qa": { - "count": null - }, - "stem_reasoning": { - "count": null, - "threshold": 80 - }, - "knowledge_qa": { - "count": 10000 - }, - "multiturn_qa": { - "count": 10000 - }, - "narrative_grounded": { - "count": 6000 - }, - "opinion_qa": { - "count": 6000 - }, - "composition_qa": { - "count": 6000 - }, - "verse_qa": { - "count": 3000 - }, - "narrative_fiction": { - "count": 3000 - } - }, - "calibration_qa": { - "count": null - }, - "authentic": { - "count": 12257 - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 200, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c2", - "reasoning-forward" - ] - }, - "config_fingerprint": "74c02b2162b87713", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2" -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/run.json deleted file mode 100644 index e39a03236040cea60a7ad4da9efec06c7a25d013..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2", - "stage": "sft", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 8352, - "branch_parent_step": null, - "config_fingerprint": "74c02b2162b87713", - "wandb_run_id": "69545234", - "created_at": 1786490713 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/meta_000082.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/meta_000082.json deleted file mode 100644 index 736c1e3f483cb3c1f9ac2f22cd214ff43a99707a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/meta_000082.json +++ /dev/null @@ -1,151 +0,0 @@ -{ - "step": 82, - "training_complete": true, - "val_bpb": 0.7636363678276777, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3", - "wandb_run_id": "1720db37", - "wandb_group": "think-d12", - "wandb_tags": "sft,curriculum,c3,scale-max,staged", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "base_step": 8352, - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "resume_from_step": null, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/config.json", - "parent_cumulative_flops": 4.5291244139996774e+19, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "59ab2594ad6fe85e4294aada37a6f0a0284eebda", - "load_optimizer": 0, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.03, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": 400, - "eval_tokens": 20971520, - "chatcore_every": 100000, - "chatcore_max_cat": -1, - "chatcore_max_sample": 32, - "save_every": -1, - "recipe": "curriculum", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c3", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C3", - "mode": "staged", - "threshold_default": 80, - "stages": [ - { - "routes": [ - "knowledge_qa" - ], - "authentic": "single" - }, - { - "routes": [ - "reasoning_qa", - "stem_reasoning", - "how_to_qa", - "opinion_qa", - "composition_qa", - "verse_qa" - ], - "calibration_qa": true - }, - { - "routes": [ - "multiturn_qa", - "narrative_grounded", - "narrative_fiction" - ], - "authentic": "multi" - } - ] - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 400, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c3", - "scale-max", - "staged" - ] - }, - "config_fingerprint": "a86d67a1560d2efa", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a86d67a1560d2efa" - }, - "loop_state": { - "step": 82, - "total_training_time": 3557.9933972358704, - "min_val_bpb": 0.7636363678276777, - "smooth_train_loss": 2.003534360747555, - "mfu": 35.111241842899126, - "tok_per_sec": 21182, - "stage_training_flops": 4.446697820258304e+17, - "inherited_parent_flops": 4.5291244139996774e+19, - "cumulative_pipeline_training_flops": 4.5735913922022605e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/model_000082.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/model_000082.pt deleted file mode 100644 index 23ef1542c884657f8f827f89460ab8b640e4ea57..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/model_000082.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:559f68d8f9c396481864e571c147c339a6858e492a5edfc1810e7c9efae149dd -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/optim_000082_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/optim_000082_rank0.pt deleted file mode 100644 index 89a93a505d07b7d1966fb19879690d3554fa5e30..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/optim_000082_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1a05a322ddaf93293b6c9fdd335cdb48104190ac2368a050f41a37fa6c49c4b7 -size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/config.json deleted file mode 100644 index 6c6dc403e5ed750bd2664a1c8849e491943208af..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/config.json +++ /dev/null @@ -1,68 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c3", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C3", - "mode": "staged", - "threshold_default": 80, - "stages": [ - { - "routes": [ - "knowledge_qa" - ], - "authentic": "single" - }, - { - "routes": [ - "reasoning_qa", - "stem_reasoning", - "how_to_qa", - "opinion_qa", - "composition_qa", - "verse_qa" - ], - "calibration_qa": true - }, - { - "routes": [ - "multiturn_qa", - "narrative_grounded", - "narrative_fiction" - ], - "authentic": "multi" - } - ] - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 400, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c3", - "scale-max", - "staged" - ] - }, - "config_fingerprint": "a86d67a1560d2efa", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3" -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/run.json deleted file mode 100644 index 8f70cb389b3fab0520d0864478ec47bfccce2a2e..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3", - "stage": "sft", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 8352, - "branch_parent_step": null, - "config_fingerprint": "a86d67a1560d2efa", - "wandb_run_id": "1720db37", - "created_at": 1786499635 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/meta_000030.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/meta_000030.json deleted file mode 100644 index bf9f8f9b84ec407ddbd87e7a206898f0129ae4b9..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/meta_000030.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 30, - "training_complete": true, - "val_bpb": 0.8584271178670264, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4", - "wandb_run_id": "4dd1e026", - "wandb_group": "think-d12", - "wandb_tags": "sft,curriculum,c4,token-balanced", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "base_step": 8352, - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "resume_from_step": null, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/config.json", - "parent_cumulative_flops": 4.5291244139996774e+19, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "59ab2594ad6fe85e4294aada37a6f0a0284eebda", - "load_optimizer": 0, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.03, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": 200, - "eval_tokens": 20971520, - "chatcore_every": 100000, - "chatcore_max_cat": -1, - "chatcore_max_sample": 32, - "save_every": -1, - "recipe": "curriculum", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c4", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C4", - "mode": "flat", - "epochs": 3, - "threshold_default": 90, - "_note": "counts pre-derived to ~4M chars (~1M tokens) per route (token-balanced)", - "routes": { - "knowledge_qa": { - "count": 13150 - }, - "opinion_qa": { - "count": 10950 - }, - "how_to_qa": { - "count": 7450 - }, - "verse_qa": { - "count": 5530 - }, - "reasoning_qa": { - "count": 4880 - }, - "multiturn_qa": { - "count": 4820 - }, - "narrative_grounded": { - "count": 4460 - }, - "composition_qa": { - "count": 4330 - }, - "stem_reasoning": { - "count": 3800 - }, - "narrative_fiction": { - "count": 2840 - } - }, - "calibration_qa": { - "count": null - }, - "authentic": { - "count": 12257 - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 200, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c4", - "token-balanced" - ] - }, - "config_fingerprint": "331c13b7908a3efc", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "331c13b7908a3efc" - }, - "loop_state": { - "step": 30, - "total_training_time": 988.5635514259338, - "min_val_bpb": 0.8463398712741395, - "smooth_train_loss": 1.813480427055826, - "mfu": 35.10521097749975, - "tok_per_sec": 21178, - "stage_training_flops": 1.62684066594816e+17, - "inherited_parent_flops": 4.5291244139996774e+19, - "cumulative_pipeline_training_flops": 4.545392820659159e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/model_000030.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/model_000030.pt deleted file mode 100644 index b3146523722931d756a499498c0e59ac90b6a1e5..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/model_000030.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:05c335bd1d3b1d0a09dcb9498c6c4bd6c0ba1f1170b979f0ccc87348dbc61fa5 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/optim_000030_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/optim_000030_rank0.pt deleted file mode 100644 index bc6ce0d089955e19eb0606a419ae40dd003b2d23..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/optim_000030_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4ab2b8e11a75e45dbb12419ce73d08f62d91dd5e3fe092984f6eb26f020f0ceb -size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/config.json deleted file mode 100644 index 29644571da19f093a9f0a829984dc2b4c7844886..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/config.json +++ /dev/null @@ -1,80 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c4", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C4", - "mode": "flat", - "epochs": 3, - "threshold_default": 90, - "_note": "counts pre-derived to ~4M chars (~1M tokens) per route (token-balanced)", - "routes": { - "knowledge_qa": { - "count": 13150 - }, - "opinion_qa": { - "count": 10950 - }, - "how_to_qa": { - "count": 7450 - }, - "verse_qa": { - "count": 5530 - }, - "reasoning_qa": { - "count": 4880 - }, - "multiturn_qa": { - "count": 4820 - }, - "narrative_grounded": { - "count": 4460 - }, - "composition_qa": { - "count": 4330 - }, - "stem_reasoning": { - "count": 3800 - }, - "narrative_fiction": { - "count": 2840 - } - }, - "calibration_qa": { - "count": null - }, - "authentic": { - "count": 12257 - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 200, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c4", - "token-balanced" - ] - }, - "config_fingerprint": "331c13b7908a3efc", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4" -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/run.json deleted file mode 100644 index 9d9d1637f4b76fc267f36c1139f47a8ef1553323..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4", - "stage": "sft", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 8352, - "branch_parent_step": null, - "config_fingerprint": "331c13b7908a3efc", - "wandb_run_id": "4dd1e026", - "created_at": 1786505289 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/meta_000052.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/meta_000052.json deleted file mode 100644 index 97e76add1e4077375fd32822a8e4ab36e854385f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/meta_000052.json +++ /dev/null @@ -1,163 +0,0 @@ -{ - "step": 52, - "training_complete": true, - "val_bpb": 0.8903500530127588, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 24, - "n_head": 12, - "n_kv_head": 12, - "n_embd": 1536, - "window_pattern": "SSSL" - }, - "user_config": { - "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5", - "wandb_run_id": "17210657", - "wandb_group": "think-d12", - "wandb_tags": "sft,curriculum,c5,domain-rebalanced", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", - "base_step": 8352, - "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints", - "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", - "resume_from_step": null, - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5", - "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/config.json", - "parent_cumulative_flops": 4.5291244139996774e+19, - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "git_commit_sha": "59ab2594ad6fe85e4294aada37a6f0a0284eebda", - "load_optimizer": 0, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.03, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": 200, - "eval_tokens": 20971520, - "chatcore_every": 100000, - "chatcore_max_cat": -1, - "chatcore_max_sample": 32, - "save_every": -1, - "recipe": "curriculum", - "curriculum_config": "", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "authentic_epochs": 0, - "knowledge_qa_epochs": 0, - "multiturn_qa_epochs": 0, - "reasoning_qa_epochs": 0, - "stem_reasoning_epochs": 0, - "narrative_grounded_epochs": 0, - "narrative_fiction_epochs": 0, - "opinion_qa_epochs": 0, - "how_to_qa_epochs": 0, - "verse_qa_epochs": 0, - "composition_qa_epochs": 0, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c5", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C5", - "mode": "domain_rebalanced", - "epochs": 3, - "threshold_default": 90, - "_note": "same route totals as C0; book_category flattened within each route", - "routes": { - "knowledge_qa": { - "count": 18000 - }, - "multiturn_qa": { - "count": 18000 - }, - "reasoning_qa": { - "count": 12000 - }, - "narrative_grounded": { - "count": 12000 - }, - "opinion_qa": { - "count": 10000 - }, - "composition_qa": { - "count": 10000 - }, - "how_to_qa": { - "count": 7000 - }, - "verse_qa": { - "count": 6000 - }, - "narrative_fiction": { - "count": 6000 - }, - "stem_reasoning": { - "count": 4000 - } - }, - "calibration_qa": { - "count": null - }, - "authentic": { - "count": 12257 - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 200, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c5", - "domain-rebalanced" - ] - }, - "config_fingerprint": "b1bbd0a977780ed3", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "b1bbd0a977780ed3" - }, - "loop_state": { - "step": 52, - "total_training_time": 2075.6576771736145, - "min_val_bpb": 0.8464341895781785, - "smooth_train_loss": 1.6294214116602808, - "mfu": 35.10245957312791, - "tok_per_sec": 21177, - "stage_training_flops": 2.819857154310144e+17, - "inherited_parent_flops": 4.5291244139996774e+19, - "cumulative_pipeline_training_flops": 4.557322985542779e+19 - } -} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/model_000052.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/model_000052.pt deleted file mode 100644 index ff34b4e7a5c19b05a2e558199e3f676354a2b8aa..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/model_000052.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6ec260000ca50247d558428abdf8630b5dd5f568e0b635ba803646617a82fee1 -size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/optim_000052_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/optim_000052_rank0.pt deleted file mode 100644 index 27850ea21de6c5714e3291ca359ad80d8aceff09..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/optim_000052_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5df00f34d178cf428a8fdfcb42733d31195ea174cf37d3b45e6106b13360d476 -size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/config.json deleted file mode 100644 index eedd423d3c5c599507d4fcf9ecb94e300b0963b9..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/config.json +++ /dev/null @@ -1,80 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-curriculum-c5", - "data": { - "recipe": "curriculum", - "curriculum": { - "name": "C5", - "mode": "domain_rebalanced", - "epochs": 3, - "threshold_default": 90, - "_note": "same route totals as C0; book_category flattened within each route", - "routes": { - "knowledge_qa": { - "count": 18000 - }, - "multiturn_qa": { - "count": 18000 - }, - "reasoning_qa": { - "count": 12000 - }, - "narrative_grounded": { - "count": 12000 - }, - "opinion_qa": { - "count": 10000 - }, - "composition_qa": { - "count": 10000 - }, - "how_to_qa": { - "count": 7000 - }, - "verse_qa": { - "count": 6000 - }, - "narrative_fiction": { - "count": 6000 - }, - "stem_reasoning": { - "count": 4000 - } - }, - "calibration_qa": { - "count": null - }, - "authentic": { - "count": 12257 - } - } - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "device_batch_size": 8, - "warmup_ratio": 0.03, - "eval_every": 200, - "chatcore_every": 100000, - "chatcore_max_sample": 32, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "curriculum", - "c5", - "domain-rebalanced" - ] - }, - "config_fingerprint": "b1bbd0a977780ed3", - "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5" -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/run.json deleted file mode 100644 index c92e891658640a2ad96ea69840d3893d2def0a1f..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5", - "stage": "sft", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_checkpoint_step": 8352, - "branch_parent_step": null, - "config_fingerprint": "b1bbd0a977780ed3", - "wandb_run_id": "17210657", - "created_at": 1786508150 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/summary.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/summary.json deleted file mode 100644 index fa27f8b246994f1fd96b1fead4b04e441cad27cb..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/summary.json +++ /dev/null @@ -1,92 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "stage": "base", - "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean-1930s", - "dataset_revision": "main", - "step": 8352, - "depth": 24, - "target_param_data_ratio": 12, - "training_tokens": 8757706752, - "final_sampled_val_bpb": 0.9204610080723857, - "minimum_sampled_val_bpb": 0.9204610080723857, - "full_val_bpb": null, - "core_metric": 0.13542864491271994, - "centered_results": { - "hellaswag_zeroshot": 0.09818760553995769, - "jeopardy": 0.0037789323832839727, - "bigbench_qa_wikidata": 0.2516116201877594, - "arc_easy": 0.22334452470143637, - "arc_challenge": -0.013651887575785318, - "copa": 0.2200000286102295, - "commonsense_qa": 0.13390662521123883, - "piqa": 0.16104459762573242, - "openbook_qa": 0.010666688283284506, - "lambada_openai": 0.35338637232780457, - "hellaswag": 0.105490247408549, - "winograd": 0.34065937995910645, - "winogrande": -0.011839032173156738, - "bigbench_dyck_languages": 0.12700000405311584, - "agi_eval_lsat_ar": 0.07608695328235625, - "bigbench_cs_algorithms": 0.38333332538604736, - "bigbench_operators": 0.11428572237491608, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.114096499979496, - "coqa": 0.16610296070575714, - "boolq": -0.060679241230613294, - "bigbench_language_identification": 0.1826182610393226 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the city of Paris, which is the capital of France. The capital of France" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is Au, and the symbol of silver is Ag. The symbol of gold is Au" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If yesterday was Friday, then tomorrow will be Saturday. If yesterday was" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is cold.\n\nThe opposite of cold is hot.\n\nThe opposite of hot is cold.\n\n" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: Mercury, Venus, Earth, Mars, Jupiter, Saturn, Uranus," - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a dark brown, with a tinge of red. It is a very pretty color" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is 13, and the equation is 13x + 3 = 13" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nTHIS essay was contributed to the \"Psychological Bulletin,\" of the University of Chicago, during the past summer; the temporary arrangements for its printing prevented its meeting the demand for it. Fair play in the matter of results in the revision of Shakespeare's plays has not been given in this field in this company, Dunning, Uzziel, and others engaging to contribute whatever they have to say on the revision and construction of even a very few plays, and the magazine enterprise to furnish more plays of a classic nature, although comprising many dramas of merit and of the utmost possible variety, is by no means of sufficient scope and interest", - "<|bos|>\n\nARTON LIBRARY OF THE DEZA SUPERINTENDENTS AND BOOKKEEPERS\n\nThe following Outline of Business\n\nthe State or Territory in which\n\nFoss work is to be carried on.\n\nPLAN OF BUSINESS Department 1. Traveling\n\n2. Records, coupons, checks, mercantile accounts, vouchers, &c.\n\n3. General office\n\n4. Subsidiary agencies. Persons dealing with department men.\n\nPlanwork.\n\nPlan.\n\n1. Plans for the method of review of machinery, etc. Uncertainty.\n\n2. Plans for securing uniformity of practice over a large field", - "<|bos|>!\n\nIt's the coss is mighty wonderous weak Thefe minutes yhere hain marry teef The tryfector\n\nOf my fandd in Life who ought to hast The pist o' a President\n\nApril 10. J,\"\n\nAmong the Arms in cison\n\nThe Arms of the President *The Arms of Washington\n\nOn October 5, (?) in another and more noticeable storm-swept plagiarism, the President bore the baptismal emblem of the United States; on the way to Washington, January 1, 1799, occurred the cancel that has made possible the interpolations of the following", - "<|bos|> ministered unto Him with great joy.\"\n\nXX. 1. CHIRIT OF THE BLESSED.\n\nThe full contrast to this joy is the hunger that followed when He left His throne in heaven, ascended, and took the place of the \"Name\" lifted up on Calvary.\n\n2. The Fulfilment of the Tabernacle: In our imagination let 50 Enter: first, the Church universal: prayer, glorification, rejoicing, nothing but an eternal army and the eternal joy which forms one side of this-and well! for consequently this is that infant Church which a voice descended, and the Redeemer Sunday broke through the wall that", - "<|bos|>eed\n\nNew York paid\n\nEach separate $1.25 and i piece by akely ried in Watts averaged 75 cents.\n\nRetail price 50 cents-No longer find it worth while to order second-hand books. Already considerable order great publishing interests are sending agents to England or Scotland to our mmission depot expressly to collect the $1.25 per set, on which income is made over to us. The influens can gain fifteen per cent, and this can be taken care of at once as it is here almost never debited. Its great advantages that has the most of A. L. Burlingame's", - "<|bos|>PREFACE\n\nWHEN PERSHAD was about to issue and was being attacked by the fanatical cuttlefish of Northern India, some weeks before at least the Calcutta of 1862 stood face to face with a panic-crazed gale, the doctors began one day on their platform to advise the world how they were going to appease the spirit of the Fish-god. The Elders and the Lady-Seconds criticized emphatically the pros and cons, and the whole city stared.\n\nOne man said: \"I am Oriental; the people of this English country do not believe in all this agnostic nonsense, and the skies when they are", - "<|bos|>iman have published,\n\nTo be completed in about 100 Monthly Parts, each illustrated by a Steel Engraving, a Work consisting of a Library of Original 180 Engravings, 3 vols. 8vo. \u00a35. 5s. strongly bound in cloth, as a Work for convenient\n\nSale by STEEL Engraved by W. L. STEENERSON, each neatly illustrated by a Steel Engraving; the Plates are in a very high state of\n\nPerfection, and warranted to give a correct Image of each Novel.\n\nThe Work is published in Monthly Parts to contain Historical and Biographical Sketches of the principal", - "<|bos|>MYSTER LIBRARY.\n\n3 2044 009 603 166 fidh\n\nQUE\n\n7808, W" - ], - "training_time_seconds": 18494.615287542343, - "stage_training_flops": 4.5291244139996774e+19, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 4.5291244139996774e+19, - "config_fingerprint": "166e7e695c1b536a", - "git_commit_sha": "60aa7005265901858554bb7ca98cd788de721e4a", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/e69c1e59", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", - "dataset_fingerprint": "0db4605cfe3a7eac", - "tokenizer_fingerprint": "21e99d5cdeeaa660", - "unique_train_tokens": 0 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/experiment_tokenizer.json deleted file mode 100644 index abb9ccfdfc6557194bfe6a3551bcdd320b777e78..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean-1930s", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 200, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 1000000000000, - "doc_cap": 1000000000, - "vocab_size": 32768 - }, - "created_at": 1784129374 -} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/token_bytes.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/token_bytes.pt deleted file mode 100644 index 737ab9ff9eafdbd5bfa971d0390b520b87ebb55a..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 -size 132649 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/tokenizer.pkl b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/tokenizer.pkl deleted file mode 100644 index 34650d2ed06bbfb645ad394f823340b08c7af1ac..0000000000000000000000000000000000000000 --- a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f -size 410542 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_000500.json b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_000500.json deleted file mode 100644 index cee5997c220788b7c64ac300d4ccaa033751f32f..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,125 +0,0 @@ -{ - "step": 500, - "experiment_id": "climbmix-d12-1epoch-25shards", - "val_bpb": 1.0398997071863882, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "climbmix-d12-1epoch-25shards", - "wandb_run_id": "2c3bf17b", - "wandb_group": "climbmix-d12", - "wandb_tags": "climbmix,d12,one-epoch,25-shards", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/base_checkpoints", - "experiment_id": "climbmix-d12-1epoch-25shards", - "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/config.json", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "climbmix-d12-1epoch-25shards", - "experiment": { - "experiment_id": "climbmix-d12-1epoch-25shards", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 25, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 1321205760, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_tokens": 1321205760, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "storage": { - "hf_model_repo": "jbduran/think-nanochat-d12", - "path": "experiments/climbmix-d12-1epoch-25shards" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "climbmix-d12-1epoch-25shards", - "group": "climbmix-d12", - "tags": [ - "climbmix", - "d12", - "one-epoch", - "25-shards" - ] - } - } - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.0398997071863882, - "smooth_train_loss": 3.4449955375416246, - "total_training_time": 1329.5420286655426 - } -} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_001000.json b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_001000.json deleted file mode 100644 index d76553c2ec68d530b6e03e3014a490340e71a635..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,125 +0,0 @@ -{ - "step": 1000, - "experiment_id": "climbmix-d12-1epoch-25shards", - "val_bpb": 0.9820657988478136, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "climbmix-d12-1epoch-25shards", - "wandb_run_id": "2c3bf17b", - "wandb_group": "climbmix-d12", - "wandb_tags": "climbmix,d12,one-epoch,25-shards", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/base_checkpoints", - "experiment_id": "climbmix-d12-1epoch-25shards", - "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/config.json", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "climbmix-d12-1epoch-25shards", - "experiment": { - "experiment_id": "climbmix-d12-1epoch-25shards", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 25, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 1321205760, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_tokens": 1321205760, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "storage": { - "hf_model_repo": "jbduran/think-nanochat-d12", - "path": "experiments/climbmix-d12-1epoch-25shards" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "climbmix-d12-1epoch-25shards", - "group": "climbmix-d12", - "tags": [ - "climbmix", - "d12", - "one-epoch", - "25-shards" - ] - } - } - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 0.9820657988478136, - "smooth_train_loss": 3.200258021225873, - "total_training_time": 2687.6018402576447 - } -} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_001500.json b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_001500.json deleted file mode 100644 index f009fe6c39fbc093881dc98f3c16d984c8097cc8..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,125 +0,0 @@ -{ - "step": 1500, - "experiment_id": "climbmix-d12-1epoch-25shards", - "val_bpb": 0.9363922843394795, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "climbmix-d12-1epoch-25shards", - "wandb_run_id": "2c3bf17b", - "wandb_group": "climbmix-d12", - "wandb_tags": "climbmix,d12,one-epoch,25-shards", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/base_checkpoints", - "experiment_id": "climbmix-d12-1epoch-25shards", - "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/config.json", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "climbmix-d12-1epoch-25shards", - "experiment": { - "experiment_id": "climbmix-d12-1epoch-25shards", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 25, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 1321205760, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_tokens": 1321205760, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "storage": { - "hf_model_repo": "jbduran/think-nanochat-d12", - "path": "experiments/climbmix-d12-1epoch-25shards" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "climbmix-d12-1epoch-25shards", - "group": "climbmix-d12", - "tags": [ - "climbmix", - "d12", - "one-epoch", - "25-shards" - ] - } - } - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 0.9363922843394795, - "smooth_train_loss": 3.0161736783997206, - "total_training_time": 4045.9229834079742 - } -} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002000.json b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002000.json deleted file mode 100644 index 162f7e88e0269237bca53ef8b693b15faac20b3d..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,125 +0,0 @@ -{ - "step": 2000, - "experiment_id": "climbmix-d12-1epoch-25shards", - "val_bpb": 0.9025021880109514, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "climbmix-d12-1epoch-25shards", - "wandb_run_id": "2c3bf17b", - "wandb_group": "climbmix-d12", - "wandb_tags": "climbmix,d12,one-epoch,25-shards", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/base_checkpoints", - "experiment_id": "climbmix-d12-1epoch-25shards", - "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/config.json", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "climbmix-d12-1epoch-25shards", - "experiment": { - "experiment_id": "climbmix-d12-1epoch-25shards", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 25, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 1321205760, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_tokens": 1321205760, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "storage": { - "hf_model_repo": "jbduran/think-nanochat-d12", - "path": "experiments/climbmix-d12-1epoch-25shards" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "climbmix-d12-1epoch-25shards", - "group": "climbmix-d12", - "tags": [ - "climbmix", - "d12", - "one-epoch", - "25-shards" - ] - } - } - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 0.9025021880109514, - "smooth_train_loss": 2.928742685399829, - "total_training_time": 5404.63369512558 - } -} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002500.json b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002500.json deleted file mode 100644 index c50d8c79a02fada4d83a52c16e2fa36c21604a07..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002500.json +++ /dev/null @@ -1,125 +0,0 @@ -{ - "step": 2500, - "experiment_id": "climbmix-d12-1epoch-25shards", - "val_bpb": 0.8791912820900084, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "climbmix-d12-1epoch-25shards", - "wandb_run_id": "2c3bf17b", - "wandb_group": "climbmix-d12", - "wandb_tags": "climbmix,d12,one-epoch,25-shards", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/base_checkpoints", - "experiment_id": "climbmix-d12-1epoch-25shards", - "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/config.json", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "climbmix-d12-1epoch-25shards", - "experiment": { - "experiment_id": "climbmix-d12-1epoch-25shards", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 25, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 1321205760, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_tokens": 1321205760, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "storage": { - "hf_model_repo": "jbduran/think-nanochat-d12", - "path": "experiments/climbmix-d12-1epoch-25shards" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "climbmix-d12-1epoch-25shards", - "group": "climbmix-d12", - "tags": [ - "climbmix", - "d12", - "one-epoch", - "25-shards" - ] - } - } - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 13, - "pos": 10792769, - "epoch": 1, - "pq_idx": 13, - "rg_idx": 10792769 - }, - "loop_state": { - "min_val_bpb": 0.8791912820900084, - "smooth_train_loss": 2.873548251177944, - "total_training_time": 6763.973432302475 - } -} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002520.json b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002520.json deleted file mode 100644 index 7892b69cc8482988c3ae57a1ed1fe8162e7b4494..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002520.json +++ /dev/null @@ -1,125 +0,0 @@ -{ - "step": 2520, - "experiment_id": "climbmix-d12-1epoch-25shards", - "val_bpb": 0.8786508624512739, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "climbmix-d12-1epoch-25shards", - "wandb_run_id": "2c3bf17b", - "wandb_group": "climbmix-d12", - "wandb_tags": "climbmix,d12,one-epoch,25-shards", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 2520, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/base_checkpoints", - "experiment_id": "climbmix-d12-1epoch-25shards", - "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/config.json", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "climbmix-d12-1epoch-25shards", - "experiment": { - "experiment_id": "climbmix-d12-1epoch-25shards", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 25, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 1321205760, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_tokens": 1321205760, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "storage": { - "hf_model_repo": "jbduran/think-nanochat-d12", - "path": "experiments/climbmix-d12-1epoch-25shards" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "climbmix-d12-1epoch-25shards", - "group": "climbmix-d12", - "tags": [ - "climbmix", - "d12", - "one-epoch", - "25-shards" - ] - } - } - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 0, - "pos": 73089, - "epoch": 2, - "pq_idx": 0, - "rg_idx": 73089 - }, - "loop_state": { - "min_val_bpb": 0.8786508624512739, - "smooth_train_loss": 2.898973586577961, - "total_training_time": 6818.212191104889 - } -} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_000500.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_000500.pt deleted file mode 100644 index 8b4308c61e02831752bfe146b268984a5ac96ae7..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f7f423d2e924189fcdee1dcf38d7bb7f412de8483b75a83ee20c3c43ef9008d6 -size 792761690 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_001000.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_001000.pt deleted file mode 100644 index b86444a6a1a21d37e802c1b9f1a3db7ffd30c35a..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c7860a7e2b7eeea2d597a78eca2e6249523c78aa393059923a7c904c89a78fc8 -size 792761690 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_001500.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_001500.pt deleted file mode 100644 index 6191ba9da5c9378c4cc43d1dd47edb693337a8f6..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:21f3a4a8eab9fda9eca2832714157828c929b991bf7494e663b2c18128a2905b -size 792761690 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002000.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002000.pt deleted file mode 100644 index 27fc1d2f7b2b41a545767163958c587c3030839c..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:dc3ba8f7f81c33286862180d8751e8c5deb9cdc030056d79ef8cf26034dede67 -size 792761690 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002500.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002500.pt deleted file mode 100644 index 0f9286cd2601a25bd7ba8372af5b9e4a833b6e06..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:291599c9d394e9b6fed99daa0fcffc694946155e441e5cb698d2c5a196a708d6 -size 792761690 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002520.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002520.pt deleted file mode 100644 index 5d5894d1fa0006feff90c939534c63d095f314ab..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002520.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fcb5db6082a7323774d8d064f9bce1cc8f8fcfd0618a855c880c4e8ca20e7ab2 -size 792761690 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_000500_rank0.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 9ea835127ef2d0b6a2187a6548f5ba35cc57cfff..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:121dc7fb80dd11ee9e3d9057cc91f2705b2395ec453ee1297675cb4023e8f094 -size 1246165357 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_001000_rank0.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 7cb264f8b5f73b95a14a009f8a95a9f26c015b52..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:184fcc2ff50f7c1504d124edea40ff56dc60cda08bcd0c4ac979c6e8ff479de1 -size 1246165357 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_001500_rank0.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 6abdd2b9a3af65234e29029223a58b1b1db1c00b..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e05e4efc73a6107d2fdf741f51145f57c3bb4ad138cc65ee8f429592aa12b7b9 -size 1246165357 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002000_rank0.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 6c9551719ebc7009e6cecb7b383e0a51df3cc3b9..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fdfcf95d93d06395de0fe12cce389019ba3c7c98c7472ef3c4ecdfc74b78dea3 -size 1246165357 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002500_rank0.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002500_rank0.pt deleted file mode 100644 index dbfea86af94ddd8c5827c7ce1b991e07aef2ad91..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:13a6c3762c33d08005feb0b07e66c6b24381738d20bee44ef2b5378d802530ff -size 1246165357 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002520_rank0.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002520_rank0.pt deleted file mode 100644 index fbca8358dd58220013a920ad4ac26dca9f0a0940..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002520_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:eac3b8631753867c67316583d315c42259fb50914e17a2bbbe84dcb3bf3162cf -size 1246165357 diff --git a/experiments/climbmix-d12-1epoch-25shards/config.json b/experiments/climbmix-d12-1epoch-25shards/config.json deleted file mode 100644 index 71d511d1ab568741433db485d0fbe874a7f269a2..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/config.json +++ /dev/null @@ -1,50 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "climbmix-d12-1epoch-25shards", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 25, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 1321205760, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_tokens": 1321205760, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "climbmix-d12-1epoch-25shards", - "group": "climbmix-d12", - "tags": ["climbmix", "d12", "one-epoch", "25-shards"] - } -} diff --git a/experiments/climbmix-d12-1epoch-25shards/evals/core.json b/experiments/climbmix-d12-1epoch-25shards/evals/core.json deleted file mode 100644 index f43d6b4aeb759b2d9b6ad355616199e0b76c931e..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/evals/core.json +++ /dev/null @@ -1,54 +0,0 @@ -{ - "model": "base_model (step 2520)", - "step": 2520, - "bpb": {}, - "core_metric": 0.1479302855639536, - "core_results": { - "hellaswag_zeroshot": 0.35480979084968567, - "jeopardy": 0.009447330608963966, - "bigbench_qa_wikidata": 0.26726046204566956, - "arc_easy": 0.563973069190979, - "arc_challenge": 0.26365187764167786, - "copa": 0.550000011920929, - "commonsense_qa": 0.3628173768520355, - "piqa": 0.6599564552307129, - "openbook_qa": 0.30400002002716064, - "lambada_openai": 0.31554433703422546, - "hellaswag": 0.34833696484565735, - "winograd": 0.5494505763053894, - "winogrande": 0.5106551051139832, - "bigbench_dyck_languages": 0.0650000050663948, - "agi_eval_lsat_ar": 0.2869565188884735, - "bigbench_cs_algorithms": 0.426515132188797, - "bigbench_operators": 0.10476190596818924, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.14692525565624237, - "coqa": 0.17825378477573395, - "boolq": 0.593883752822876, - "bigbench_language_identification": 0.2522999942302704 - }, - "centered_results": { - "hellaswag_zeroshot": 0.1397463877995809, - "jeopardy": 0.009447330608963966, - "bigbench_qa_wikidata": 0.26726046204566956, - "arc_easy": 0.41863075892130536, - "arc_challenge": 0.01820250352223714, - "copa": 0.10000002384185791, - "commonsense_qa": 0.20352172106504438, - "piqa": 0.3199129104614258, - "openbook_qa": 0.07200002670288086, - "lambada_openai": 0.31554433703422546, - "hellaswag": 0.13111595312754312, - "winograd": 0.09890115261077881, - "winogrande": 0.02131021022796631, - "bigbench_dyck_languages": 0.0650000050663948, - "agi_eval_lsat_ar": 0.10869564861059187, - "bigbench_cs_algorithms": 0.426515132188797, - "bigbench_operators": 0.10476190596818924, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.14692525565624237, - "coqa": 0.17825378477573395, - "boolq": -0.06872696625558952, - "bigbench_language_identification": 0.17744773842714012 - } -} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/evals/val_bpb.json b/experiments/climbmix-d12-1epoch-25shards/evals/val_bpb.json deleted file mode 100644 index 997556146e0dd714c7a659dc441b65ac486cfa8e..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/evals/val_bpb.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "model": "base_model (step 2520)", - "step": 2520, - "bpb": { - "val": 0.8792611250626658 - }, - "core_metric": null, - "core_results": null, - "centered_results": null -} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/run.json b/experiments/climbmix-d12-1epoch-25shards/run.json deleted file mode 100644 index 07ce2daa96cfbab3d52418d1bd373cbd82662bdf..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/run.json +++ /dev/null @@ -1,5 +0,0 @@ -{ - "experiment_id": "climbmix-d12-1epoch-25shards", - "wandb_run_id": "2c3bf17b", - "created_at": 1781142948 -} diff --git a/experiments/climbmix-d12-1epoch-25shards/summary.json b/experiments/climbmix-d12-1epoch-25shards/summary.json deleted file mode 100644 index 070a73bcf0e42cdac605b07d7597d026dbd14301..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/summary.json +++ /dev/null @@ -1,44 +0,0 @@ -{ - "experiment_id": "climbmix-d12-1epoch-25shards", - "dataset": "karpathy/climbmix-400b-shuffle", - "dataset_revision": "main", - "step": 2520, - "depth": 12, - "target_param_data_ratio": null, - "training_tokens": 1321205760, - "final_sampled_val_bpb": 0.8786508624512739, - "minimum_sampled_val_bpb": 0.8786508624512739, - "full_val_bpb": 0.8792611250626658, - "core_metric": 0.1479302855639536, - "centered_results": { - "hellaswag_zeroshot": 0.1397463877995809, - "jeopardy": 0.009447330608963966, - "bigbench_qa_wikidata": 0.26726046204566956, - "arc_easy": 0.41863075892130536, - "arc_challenge": 0.01820250352223714, - "copa": 0.10000002384185791, - "commonsense_qa": 0.20352172106504438, - "piqa": 0.3199129104614258, - "openbook_qa": 0.07200002670288086, - "lambada_openai": 0.31554433703422546, - "hellaswag": 0.13111595312754312, - "winograd": 0.09890115261077881, - "winogrande": 0.02131021022796631, - "bigbench_dyck_languages": 0.0650000050663948, - "agi_eval_lsat_ar": 0.10869564861059187, - "bigbench_cs_algorithms": 0.426515132188797, - "bigbench_operators": 0.10476190596818924, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.14692525565624237, - "coqa": 0.17825378477573395, - "boolq": -0.06872696625558952, - "bigbench_language_identification": 0.17744773842714012 - }, - "training_time_seconds": 6818.212191104889, - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/2c3bf17b", - "huggingface_url": "https://huggingface.co/jbduran/think-nanochat-d12/tree/main/experiments/climbmix-d12-1epoch-25shards", - "dataset_fingerprint": "29713846c3c08835", - "tokenizer_fingerprint": "285fd2719d7380ad", - "unique_train_tokens": 1321205760, - "effective_epochs": 1.0 -} diff --git a/experiments/climbmix-d12-1epoch-25shards/tokenizer/experiment_tokenizer.json b/experiments/climbmix-d12-1epoch-25shards/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 5c6b16fa95137e24b9355f413e2659d44189a44c..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,19 +0,0 @@ -{ - "experiment_id": "climbmix-d12-1epoch-25shards", - "dataset": { - "adapter": "parquet_shards", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", - "validation_shard": 6542, - "num_train_shards": 25, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1781143229 -} diff --git a/experiments/climbmix-d12-1epoch-25shards/tokenizer/token_bytes.pt b/experiments/climbmix-d12-1epoch-25shards/tokenizer/token_bytes.pt deleted file mode 100644 index 9b3435c49ec7d41906a23e6147e57ff3fc94020d..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:009ef93d20dd4684497c19c4b7fed29278b53f20022d2dc39264e27b9eefa2a8 -size 132649 diff --git a/experiments/climbmix-d12-1epoch-25shards/tokenizer/tokenizer.pkl b/experiments/climbmix-d12-1epoch-25shards/tokenizer/tokenizer.pkl deleted file mode 100644 index 019b5ce0de46ab6fb98b9b3ed0956aeed483fd44..0000000000000000000000000000000000000000 --- a/experiments/climbmix-d12-1epoch-25shards/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:ae73c5f7a960edc56022dfd46a653df2b9b38d84456e5e9f48eb5e02a60c21c2 -size 412126 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_000500.json b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_000500.json deleted file mode 100644 index 842f9e37a0db08c5ba32b6a13ac91bafb29a6a71..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,186 +0,0 @@ -{ - "step": 500, - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "val_bpb": 1.3171868025813407, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "ib80climb20-d12-1ep-26sh-r12", - "wandb_run_id": "cc999c41", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,climbmix,d12,1ep,26sh,r12,mix-80-20", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "experiment_config": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/config.json", - "tokenizer_fingerprint": "ff974929ec9d57e4", - "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "ib80climb20-d12-1ep-26sh-r12", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "datasets": [ - { - "name": "institutional-books", - "repo": "jbduran/think-dataset", - "revision": "main", - "ratio": 0.8, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7, - 8, - 9, - 10, - 11, - 12, - 13, - 14, - 15, - 16, - 17, - 18, - 19, - 20, - 21, - 22 - ], - "validation_shard": 472, - "download_workers": 4 - }, - { - "name": "climbmix", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "ratio": 0.2, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7 - ], - "validation_shard": 99, - "download_workers": 4 - } - ], - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "ib80climb20-d12-1ep-26sh-r12", - "group": "think-d12", - "tags": [ - "think-dataset", - "climbmix", - "d12", - "1ep", - "26sh", - "r12", - "mix-80-20" - ] - }, - "config_fingerprint": "fada94714cc2dbff", - "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" - }, - "stage": "base", - "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fada94714cc2dbff" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3171868025813407, - "smooth_train_loss": 3.7414761193419888, - "total_training_time": 1282.0482211112976, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_001000.json b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_001000.json deleted file mode 100644 index 345d6bc7b2870f71bd825acbcad144caa0ab7dcf..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,186 +0,0 @@ -{ - "step": 1000, - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "val_bpb": 1.2342483403874234, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "ib80climb20-d12-1ep-26sh-r12", - "wandb_run_id": "cc999c41", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,climbmix,d12,1ep,26sh,r12,mix-80-20", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "experiment_config": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/config.json", - "tokenizer_fingerprint": "ff974929ec9d57e4", - "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "ib80climb20-d12-1ep-26sh-r12", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "datasets": [ - { - "name": "institutional-books", - "repo": "jbduran/think-dataset", - "revision": "main", - "ratio": 0.8, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7, - 8, - 9, - 10, - 11, - 12, - 13, - 14, - 15, - 16, - 17, - 18, - 19, - 20, - 21, - 22 - ], - "validation_shard": 472, - "download_workers": 4 - }, - { - "name": "climbmix", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "ratio": 0.2, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7 - ], - "validation_shard": 99, - "download_workers": 4 - } - ], - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "ib80climb20-d12-1ep-26sh-r12", - "group": "think-d12", - "tags": [ - "think-dataset", - "climbmix", - "d12", - "1ep", - "26sh", - "r12", - "mix-80-20" - ] - }, - "config_fingerprint": "fada94714cc2dbff", - "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" - }, - "stage": "base", - "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fada94714cc2dbff" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2342483403874234, - "smooth_train_loss": 3.5135694958986257, - "total_training_time": 2587.891883611679, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_001500.json b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_001500.json deleted file mode 100644 index 66ba21ca84abfe2818431eba6bb5e7722c9c061e..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,186 +0,0 @@ -{ - "step": 1500, - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "val_bpb": 1.1748804664299872, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "ib80climb20-d12-1ep-26sh-r12", - "wandb_run_id": "cc999c41", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,climbmix,d12,1ep,26sh,r12,mix-80-20", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "experiment_config": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/config.json", - "tokenizer_fingerprint": "ff974929ec9d57e4", - "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "ib80climb20-d12-1ep-26sh-r12", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "datasets": [ - { - "name": "institutional-books", - "repo": "jbduran/think-dataset", - "revision": "main", - "ratio": 0.8, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7, - 8, - 9, - 10, - 11, - 12, - 13, - 14, - 15, - 16, - 17, - 18, - 19, - 20, - 21, - 22 - ], - "validation_shard": 472, - "download_workers": 4 - }, - { - "name": "climbmix", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "ratio": 0.2, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7 - ], - "validation_shard": 99, - "download_workers": 4 - } - ], - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "ib80climb20-d12-1ep-26sh-r12", - "group": "think-d12", - "tags": [ - "think-dataset", - "climbmix", - "d12", - "1ep", - "26sh", - "r12", - "mix-80-20" - ] - }, - "config_fingerprint": "fada94714cc2dbff", - "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" - }, - "stage": "base", - "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fada94714cc2dbff" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.1748804664299872, - "smooth_train_loss": 3.1845675293563995, - "total_training_time": 3895.9449546337128, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002000.json b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002000.json deleted file mode 100644 index 8c9e3b1867589272693573de8798cd5758a6152a..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,186 +0,0 @@ -{ - "step": 2000, - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "val_bpb": 1.128405625513683, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "ib80climb20-d12-1ep-26sh-r12", - "wandb_run_id": "cc999c41", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,climbmix,d12,1ep,26sh,r12,mix-80-20", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "experiment_config": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/config.json", - "tokenizer_fingerprint": "ff974929ec9d57e4", - "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "ib80climb20-d12-1ep-26sh-r12", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "datasets": [ - { - "name": "institutional-books", - "repo": "jbduran/think-dataset", - "revision": "main", - "ratio": 0.8, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7, - 8, - 9, - 10, - 11, - 12, - 13, - 14, - 15, - 16, - 17, - 18, - 19, - 20, - 21, - 22 - ], - "validation_shard": 472, - "download_workers": 4 - }, - { - "name": "climbmix", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "ratio": 0.2, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7 - ], - "validation_shard": 99, - "download_workers": 4 - } - ], - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "ib80climb20-d12-1ep-26sh-r12", - "group": "think-d12", - "tags": [ - "think-dataset", - "climbmix", - "d12", - "1ep", - "26sh", - "r12", - "mix-80-20" - ] - }, - "config_fingerprint": "fada94714cc2dbff", - "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" - }, - "stage": "base", - "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fada94714cc2dbff" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.128405625513683, - "smooth_train_loss": 3.155137224481085, - "total_training_time": 5204.380163908005, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002500.json b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002500.json deleted file mode 100644 index ba830a7739906657bdc1001714fe348bd3275490..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002500.json +++ /dev/null @@ -1,186 +0,0 @@ -{ - "step": 2500, - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "val_bpb": 1.2390006511364335, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "ib80climb20-d12-1ep-26sh-r12", - "wandb_run_id": "cc999c41", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,climbmix,d12,1ep,26sh,r12,mix-80-20", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "experiment_config": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/config.json", - "tokenizer_fingerprint": "ff974929ec9d57e4", - "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "ib80climb20-d12-1ep-26sh-r12", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "datasets": [ - { - "name": "institutional-books", - "repo": "jbduran/think-dataset", - "revision": "main", - "ratio": 0.8, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7, - 8, - 9, - 10, - 11, - 12, - 13, - 14, - 15, - 16, - 17, - 18, - 19, - 20, - 21, - 22 - ], - "validation_shard": 472, - "download_workers": 4 - }, - { - "name": "climbmix", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "ratio": 0.2, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7 - ], - "validation_shard": 99, - "download_workers": 4 - } - ], - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "ib80climb20-d12-1ep-26sh-r12", - "group": "think-d12", - "tags": [ - "think-dataset", - "climbmix", - "d12", - "1ep", - "26sh", - "r12", - "mix-80-20" - ] - }, - "config_fingerprint": "fada94714cc2dbff", - "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" - }, - "stage": "base", - "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fada94714cc2dbff" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 13, - "pos": 10792769, - "epoch": 1, - "pq_idx": 13, - "rg_idx": 10792769 - }, - "loop_state": { - "min_val_bpb": 1.128405625513683, - "smooth_train_loss": 3.283352044948622, - "total_training_time": 6508.467170000076, - "stage_training_flops": 1162736943759360000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1162736943759360000 - } -} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002520.json b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002520.json deleted file mode 100644 index f4b44299b6be8a45ccdc67becfe28f7b38f13ee2..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002520.json +++ /dev/null @@ -1,186 +0,0 @@ -{ - "step": 2520, - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "val_bpb": 1.2371360397130737, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "ib80climb20-d12-1ep-26sh-r12", - "wandb_run_id": "cc999c41", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,climbmix,d12,1ep,26sh,r12,mix-80-20", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 12.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "experiment_config": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/config.json", - "tokenizer_fingerprint": "ff974929ec9d57e4", - "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "ib80climb20-d12-1ep-26sh-r12", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "datasets": [ - { - "name": "institutional-books", - "repo": "jbduran/think-dataset", - "revision": "main", - "ratio": 0.8, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7, - 8, - 9, - 10, - 11, - 12, - 13, - 14, - 15, - 16, - 17, - 18, - 19, - 20, - 21, - 22 - ], - "validation_shard": 472, - "download_workers": 4 - }, - { - "name": "climbmix", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "ratio": 0.2, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7 - ], - "validation_shard": 99, - "download_workers": 4 - } - ], - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "ib80climb20-d12-1ep-26sh-r12", - "group": "think-d12", - "tags": [ - "think-dataset", - "climbmix", - "d12", - "1ep", - "26sh", - "r12", - "mix-80-20" - ] - }, - "config_fingerprint": "fada94714cc2dbff", - "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" - }, - "stage": "base", - "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fada94714cc2dbff" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 13, - "pos": 21278849, - "epoch": 1, - "pq_idx": 13, - "rg_idx": 21278849 - }, - "loop_state": { - "min_val_bpb": 1.128405625513683, - "smooth_train_loss": 3.247208542986062, - "total_training_time": 6560.660916090012, - "stage_training_flops": 1172038839309434880, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1172038839309434880 - } -} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_000500.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_000500.pt deleted file mode 100644 index 4a6f5efe5ea389bcf4da07f147c3c15d51711df5..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0ee5c02fdfc77f8359298d79855fbdd1e3e25cfb4465aec43b504b4b330d0af1 -size 792761690 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_001000.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_001000.pt deleted file mode 100644 index 624d0c04ece9410b6d1ffe52ced48a57dda933ff..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4b6b89ed1d325c25bda441dd93e600faae27d96717a5756c3fe83add4edea20a -size 792761690 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_001500.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_001500.pt deleted file mode 100644 index c3533dbf65367af639a036dd9cd400f1a0544c6e..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1fa16ac0740c422f5867cfe6e2b312e1d9fe6f3c3093c4a562925d77973520dc -size 792761690 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002000.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002000.pt deleted file mode 100644 index e9a1da87bdfcab033f0b7700e0b0b614812b2ebb..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:80c5aefbead9aecd2f28352d73c18fac71d4f37ba129be6b2e02945206b1ea05 -size 792761690 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002500.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002500.pt deleted file mode 100644 index a03a9376adccb70fa14e69f1f928f894d65c552e..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7b5486ba7ac7781aef2d71fbf47f1e70d1d1cc2ce85583008d0f277c39c9c0d3 -size 792761690 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002520.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002520.pt deleted file mode 100644 index 74f19b5f0750c15eb7afdc0104a40d5157722e2c..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002520.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a6f5dcb68f4be9513e13c428083e504c6df437d0e6ae3cbbf0aa2651587de86c -size 792761690 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_000500_rank0.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 953613094983098446ec00174b2e62741e01a626..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:dc23a7d1d8843b5b02703b96040f548eb8b3d72ed12a7187e5a93fcbeecbea5b -size 1246165357 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_001000_rank0.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 6c2bfbc46eb4af8bdd430b416014f1944470dd7b..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:502fb4033c7a8e961223052edb7215378cadf96c0ed7a94f536f463ecc807ee6 -size 1246165357 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_001500_rank0.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 407d540d2f40a8c484940b679755abece4c11803..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:26e19a82ec567925ba5e89d354c8d4b350bdcbe5de3d1b1be049a19810517941 -size 1246165357 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002000_rank0.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 493a635e49093fdcd5fe4b6f599912fb53064168..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8d0e35bf7141fc966f63ef557afd358b9a18f3c18842f51a214a42bf471f694b -size 1246165357 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002500_rank0.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002500_rank0.pt deleted file mode 100644 index b11e7bc4a101af6a343ce1f1a006f9f27b3d51e2..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5609a688d27407fae87a216e3d13a417375068420cdd5dbe5fc21a16434de284 -size 1246165357 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002520_rank0.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002520_rank0.pt deleted file mode 100644 index 5f54ba46244e3a7eeeaf0cd48d159718ae1d64c0..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002520_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0e3a255abcbf36fb601466c2b4a87c8644a9f2c5f7fc8798097e090fc164012a -size 1246165357 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/config.json b/experiments/ib80climb20-d12-1ep-26sh-r12/config.json deleted file mode 100644 index 5b218744464f38e9c9696f4ed3efcb36634f61a0..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/config.json +++ /dev/null @@ -1,105 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "datasets": [ - { - "name": "institutional-books", - "repo": "jbduran/think-dataset", - "revision": "main", - "ratio": 0.8, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7, - 8, - 9, - 10, - 11, - 12, - 13, - 14, - 15, - 16, - 17, - 18, - 19, - 20, - 21, - 22 - ], - "validation_shard": 472, - "download_workers": 4 - }, - { - "name": "climbmix", - "repo": "karpathy/climbmix-400b-shuffle", - "revision": "main", - "ratio": 0.2, - "train_shards": [ - 0, - 1, - 2, - 3, - 4, - 5, - 6, - 7 - ], - "validation_shard": 99, - "download_workers": 4 - } - ], - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 12.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "ib80climb20-d12-1ep-26sh-r12", - "group": "think-d12", - "tags": [ - "think-dataset", - "climbmix", - "d12", - "1ep", - "26sh", - "r12", - "mix-80-20" - ] - }, - "config_fingerprint": "fada94714cc2dbff", - "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" -} diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/evals/core.json b/experiments/ib80climb20-d12-1ep-26sh-r12/evals/core.json deleted file mode 100644 index d7627ff74411749b8248c61caa081b26bf04b3d8..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2520)", - "step": 2520, - "bpb": {}, - "core_metric": 0.10026160432378077, - "core_results": { - "hellaswag_zeroshot": 0.2837084233760834, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.16987352073192596, - "arc_easy": 0.4431818127632141, - "arc_challenge": 0.24573378264904022, - "copa": 0.5, - "commonsense_qa": 0.2874692976474762, - "piqa": 0.6147986650466919, - "openbook_qa": 0.2760000228881836, - "lambada_openai": 0.29497379064559937, - "hellaswag": 0.28211510181427, - "winograd": 0.5970696210861206, - "winogrande": 0.5067087411880493, - "bigbench_dyck_languages": 0.026000000536441803, - "agi_eval_lsat_ar": 0.25652173161506653, - "bigbench_cs_algorithms": 0.415909081697464, - "bigbench_operators": 0.10000000149011612, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.13396404683589935, - "coqa": 0.10948264598846436, - "boolq": 0.5363914370536804, - "bigbench_language_identification": 0.2574999928474426 - }, - "centered_results": { - "hellaswag_zeroshot": 0.044944564501444496, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.16987352073192596, - "arc_easy": 0.25757575035095215, - "arc_challenge": -0.005688289801279704, - "copa": 0.0, - "commonsense_qa": 0.10933662205934523, - "piqa": 0.2295973300933838, - "openbook_qa": 0.03466669718424479, - "lambada_openai": 0.29497379064559937, - "hellaswag": 0.042820135752360024, - "winograd": 0.1941392421722412, - "winogrande": 0.013417482376098633, - "bigbench_dyck_languages": 0.026000000536441803, - "agi_eval_lsat_ar": 0.07065216451883315, - "bigbench_cs_algorithms": 0.415909081697464, - "bigbench_operators": 0.10000000149011612, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.13396404683589935, - "coqa": 0.10948264598846436, - "boolq": -0.22002253406926203, - "bigbench_language_identification": 0.1831683089630832 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/evals/samples.json b/experiments/ib80climb20-d12-1ep-26sh-r12/evals/samples.json deleted file mode 100644 index 4dd108d1b3a216f293c46031df4426da40cbdfcf..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2520)", - "step": 2520, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the world. The capital of the world is the capital of the" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is gold. It is a chemical element that is used to make gold. It is" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If you're not sure what to do, you can check out the" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is cold. The opposite of cold is cold. The opposite of cold is hot." - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the sun, the moon, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the red. I like the red and the green. I like the red and" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times the number of times the number of times the number of times" - } - ], - "unconditioned_samples": [ - "<|bos|>How to make a mock cryptic PR, a February 2016 article from the authors section of the \"Types of People\", can be found in the web site for more details.\n\nQuestion: What is one of the benefits of using ReinforcementAmit? Answer: ReinforcementAmit provides players with a healthy diet and strategies for successful business, since it exerts the widest selection of resources.", - "<|bos|>Chairman's Manual\n\nAdvertisement\n\nShareExplore\n\nASSISTY, MILLING DELAY\n\nEXECUTIVE INTERRANING THESE TI MINERALS DO NOT AGE\n\nJust compile a summary of the 18 endangered species of sea turtle from recently collected sea turtle remains in the estuarine and estuarine environments with the help of the National Oceanic and Atmospheric Administration (NOAA)", - "<|bos|>In my first time as a family I used to go to a lot of things to read and to get to know the basic components that allowed for healthy (good) hair and makeup for hair. I've never had it before, so it's fine to keep reading and if you top off your hair from time to time, that might be the reason.\n\nBut where I trace hair and makeup and make sure to get the right/Hair and hair profiles. And cutting the seams in the base of your chinboard is a good thing. And there are many bodices of the same breed as when you shine a little and pencil through your fingertips", - "<|bos|>Directions: 1) Fill a bucket with water in a large bowl. 2) Pour in the spices and seasonings. 3) In a bowl whisk together the diced tomatoes, onion, and garlic. 4) Place the diced bell peppers in a bowl and shake vigorously. Set aside.\n4) Introduce the diced tomatoes: Once the bell peppers have reached a firmness, they can be dropped rapidly. Place the peppers into the bowl and put a thought on them. Remember, they just never need a question.\n\nQuestion: What is the purpose of the rice balls and the peppers in the aforementioned mixture? Answer: To form", - "<|bos|>Dave's culinary debut was among the Falcons of the Snakes. It's still an easy pop-off into excess eating to the dogs, whether they're good or bad `` toggles ''. And one of the most noticeable aspects of Falcons is that they are long and tufted: one thing is for sure! Quick- And-Goopsy giving, step by step ; turning the buttons-lasts how group meal is structured.It's that easy, Step-by-step! The Falcons eat all the food together , and form the perfect organic whole - bite - part of their savory food.\nWhen Ozemne comes", - "<|bos|>Hardware\nE. Ten. Airds rayed and meshed in Annex 0 and 1\u2026\n\nExamples of PDF: Documents and Models, a comprehensive resource written and likewise clearly explained on the pieces page by an author in excellent colleagues.. Unlike part numbers, the original document reproduction is nethelix (to be more precise).", - "<|bos|>Well, you may want to know that this project also gives Oregon researchers an opportunity to learn how her stocks relate to the K-127. Three months later, you will find that the B-127 is providing access to a wealth of secrets about the K-127 that still have a long time to read.\n\nAs for how Oregon's secrets will affect her assets, you can look to the right tab below, just for a moment.\n\n1. Banejiq\n\nThis man was in Texas, while the other two brothers were UA's. What kept you from making the discovery you in the right place may be doubt", - "<|bos|>The sharing of information among grass producers gives farmers reason to believe they need to be able to handle climate change substantially better and, as part of their efforts to ensure the safety of agriculture while at the same time improving their health and fertility, it has set the tone for a research project in Prague, where it has revealed weather differences, so that the USDA beef farmer\u00b4s hours of practical knowledge have been strengthened.\n\nThough he lost farm productivity a few weeks ago, he had access to seasonal food staples like tomatoes, zucchini, cucumbers and cottage cheese and for the most part made friends with farmers from around the city. However, one year" - ] -} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/evals/val_bpb.json b/experiments/ib80climb20-d12-1ep-26sh-r12/evals/val_bpb.json deleted file mode 100644 index f2060a97326fd80f76ce2e6f89c780fc95096fa2..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2520)", - "step": 2520, - "bpb": { - "val": 1.1701973227550237 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/run.json b/experiments/ib80climb20-d12-1ep-26sh-r12/run.json deleted file mode 100644 index cf97c574583a11bc7fd8e3493a7c274c4da4707c..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "stage": "base", - "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fada94714cc2dbff", - "wandb_run_id": "cc999c41", - "created_at": 1781542858 -} diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/summary.json b/experiments/ib80climb20-d12-1ep-26sh-r12/summary.json deleted file mode 100644 index ca8ff2ffa44c2b68954b87f616b9f23b32906b9b..0000000000000000000000000000000000000000 --- a/experiments/ib80climb20-d12-1ep-26sh-r12/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "stage": "base", - "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset+karpathy/climbmix-400b-shuffle", - "dataset_revision": null, - "step": 2520, - "depth": 12, - "target_param_data_ratio": 12.0, - "training_tokens": 1321205760, - "final_sampled_val_bpb": 1.2371360397130737, - "minimum_sampled_val_bpb": 1.128405625513683, - "full_val_bpb": 1.1701973227550237, - "core_metric": 0.10026160432378077, - "centered_results": { - "hellaswag_zeroshot": 0.044944564501444496, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.16987352073192596, - "arc_easy": 0.25757575035095215, - "arc_challenge": -0.005688289801279704, - "copa": 0.0, - "commonsense_qa": 0.10933662205934523, - "piqa": 0.2295973300933838, - "openbook_qa": 0.03466669718424479, - "lambada_openai": 0.29497379064559937, - "hellaswag": 0.042820135752360024, - "winograd": 0.1941392421722412, - "winogrande": 0.013417482376098633, - "bigbench_dyck_languages": 0.026000000536441803, - "agi_eval_lsat_ar": 0.07065216451883315, - "bigbench_cs_algorithms": 0.415909081697464, - "bigbench_operators": 0.10000000149011612, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.13396404683589935, - "coqa": 0.10948264598846436, - "boolq": -0.22002253406926203, - "bigbench_language_identification": 0.1831683089630832 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the world. The capital of the world is the capital of the" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is gold. It is a chemical element that is used to make gold. It is" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If you're not sure what to do, you can check out the" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is cold. The opposite of cold is cold. The opposite of cold is hot." - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the sun, the moon, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the red. I like the red and the green. I like the red and" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times the number of times the number of times the number of times" - } - ], - "unconditioned_samples": [ - "<|bos|>How to make a mock cryptic PR, a February 2016 article from the authors section of the \"Types of People\", can be found in the web site for more details.\n\nQuestion: What is one of the benefits of using ReinforcementAmit? Answer: ReinforcementAmit provides players with a healthy diet and strategies for successful business, since it exerts the widest selection of resources.", - "<|bos|>Chairman's Manual\n\nAdvertisement\n\nShareExplore\n\nASSISTY, MILLING DELAY\n\nEXECUTIVE INTERRANING THESE TI MINERALS DO NOT AGE\n\nJust compile a summary of the 18 endangered species of sea turtle from recently collected sea turtle remains in the estuarine and estuarine environments with the help of the National Oceanic and Atmospheric Administration (NOAA)", - "<|bos|>In my first time as a family I used to go to a lot of things to read and to get to know the basic components that allowed for healthy (good) hair and makeup for hair. I've never had it before, so it's fine to keep reading and if you top off your hair from time to time, that might be the reason.\n\nBut where I trace hair and makeup and make sure to get the right/Hair and hair profiles. And cutting the seams in the base of your chinboard is a good thing. And there are many bodices of the same breed as when you shine a little and pencil through your fingertips", - "<|bos|>Directions: 1) Fill a bucket with water in a large bowl. 2) Pour in the spices and seasonings. 3) In a bowl whisk together the diced tomatoes, onion, and garlic. 4) Place the diced bell peppers in a bowl and shake vigorously. Set aside.\n4) Introduce the diced tomatoes: Once the bell peppers have reached a firmness, they can be dropped rapidly. Place the peppers into the bowl and put a thought on them. Remember, they just never need a question.\n\nQuestion: What is the purpose of the rice balls and the peppers in the aforementioned mixture? Answer: To form", - "<|bos|>Dave's culinary debut was among the Falcons of the Snakes. It's still an easy pop-off into excess eating to the dogs, whether they're good or bad `` toggles ''. And one of the most noticeable aspects of Falcons is that they are long and tufted: one thing is for sure! Quick- And-Goopsy giving, step by step ; turning the buttons-lasts how group meal is structured.It's that easy, Step-by-step! The Falcons eat all the food together , and form the perfect organic whole - bite - part of their savory food.\nWhen Ozemne comes", - "<|bos|>Hardware\nE. Ten. Airds rayed and meshed in Annex 0 and 1\u2026\n\nExamples of PDF: Documents and Models, a comprehensive resource written and likewise clearly explained on the pieces page by an author in excellent colleagues.. Unlike part numbers, the original document reproduction is nethelix (to be more precise).", - "<|bos|>Well, you may want to know that this project also gives Oregon researchers an opportunity to learn how her stocks relate to the K-127. Three months later, you will find that the B-127 is providing access to a wealth of secrets about the K-127 that still have a long time to read.\n\nAs for how Oregon's secrets will affect her assets, you can look to the right tab below, just for a moment.\n\n1. Banejiq\n\nThis man was in Texas, while the other two brothers were UA's. What kept you from making the discovery you in the right place may be doubt", - "<|bos|>The sharing of information among grass producers gives farmers reason to believe they need to be able to handle climate change substantially better and, as part of their efforts to ensure the safety of agriculture while at the same time improving their health and fertility, it has set the tone for a research project in Prague, where it has revealed weather differences, so that the USDA beef farmer\u00b4s hours of practical knowledge have been strengthened.\n\nThough he lost farm productivity a few weeks ago, he had access to seasonal food staples like tomatoes, zucchini, cucumbers and cottage cheese and for the most part made friends with farmers from around the city. However, one year" - ], - "training_time_seconds": 6560.660916090012, - "stage_training_flops": 1.172038839309435e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.172038839309435e+18, - "config_fingerprint": "fada94714cc2dbff", - "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/cc999c41", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/ib80climb20-d12-1ep-26sh-r12", - "dataset_fingerprint": "643329665ca62fe0", - "tokenizer_fingerprint": "ff974929ec9d57e4", - "unique_train_tokens": 1360841933, - "effective_epochs": 0.9708737862650798 -} diff --git a/experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1/config.json b/experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1/config.json deleted file mode 100644 index 12486bedcafce19f61c411e91cbe3757ae01875d..0000000000000000000000000000000000000000 --- a/experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1/config.json +++ /dev/null @@ -1,50 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "complete-modern-sft-v1", - "parent": { - "base_experiment_id": "karpathy-nanochat-d34", - "checkpoint_step": 169150 - }, - "data": { - "recipe": "nanochat-default", - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "max_train_presentations": -1 - }, - "training": { - "num_iterations": -1, - "load_optimizer": 0, - "max_seq_len": 2048, - "device_batch_size": 4, - "total_batch_size": 524288, - "embedding_lr": 0.2, - "unembedding_lr": 0.004, - "matrix_lr": 0.02, - "init_lr_frac": 0.8, - "warmup_ratio": 0.0, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": -1, - "chatcore_every": -1, - "save_every": 200 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "karpathy-d34", - "tags": [ - "sft", - "karpathy-d34", - "nanochat-default", - "complete-mixture", - "fresh-optimizer" - ] - }, - "config_fingerprint": "d2c5a5ecd651045d", - "artifact_path": "experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1" -} diff --git a/experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1/run.json b/experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1/run.json deleted file mode 100644 index ae64c46c7675f84afd409f065cfc57c41367cc25..0000000000000000000000000000000000000000 --- a/experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1/run.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "experiment_id": "karpathy-nanochat-d34-complete-modern-sft-v1", - "stage": "sft", - "base_experiment_id": "karpathy-nanochat-d34", - "parent_experiment_id": "karpathy-nanochat-d34", - "parent_checkpoint_step": 169150, - "branch_parent_step": null, - "config_fingerprint": "d2c5a5ecd651045d", - "wandb_run_id": "b34a37b6", - "created_at": 1787102531 -} diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_000500.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_000500.json deleted file mode 100644 index ae7bf02b4ea1eca6f729d198cc376edde5744feb..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 500, - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "val_bpb": 1.4440721103388112, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "pre1900-d12-1ep-4sh-r20", - "wandb_run_id": "59d78984", - "wandb_group": "pre1900-d12", - "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", - "tokenizer_fingerprint": "ab10cc8eda78637c", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "pre1900-d12-1ep-4sh-r20", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 4, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "pre1900-d12-1ep-4sh-r20", - "group": "pre1900-d12", - "tags": [ - "pre1900-corpus", - "d12", - "ratio20", - "4shards", - "1epoch" - ] - }, - "config_fingerprint": "27dd1d0b7b2eff44", - "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" - }, - "stage": "base", - "base_experiment_id": "pre1900-d12-1ep-4sh-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "27dd1d0b7b2eff44" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.4440721103388112, - "smooth_train_loss": 4.017187875737277, - "total_training_time": 1297.9931802749634, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_001000.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_001000.json deleted file mode 100644 index 1794dac81c418b665bf9a8357df29c84af7f6090..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 1000, - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "val_bpb": 1.3600367137774247, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "pre1900-d12-1ep-4sh-r20", - "wandb_run_id": "59d78984", - "wandb_group": "pre1900-d12", - "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", - "tokenizer_fingerprint": "ab10cc8eda78637c", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "pre1900-d12-1ep-4sh-r20", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 4, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "pre1900-d12-1ep-4sh-r20", - "group": "pre1900-d12", - "tags": [ - "pre1900-corpus", - "d12", - "ratio20", - "4shards", - "1epoch" - ] - }, - "config_fingerprint": "27dd1d0b7b2eff44", - "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" - }, - "stage": "base", - "base_experiment_id": "pre1900-d12-1ep-4sh-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "27dd1d0b7b2eff44" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.3600367137774247, - "smooth_train_loss": 3.664376154963864, - "total_training_time": 2624.0160751342773, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_001500.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_001500.json deleted file mode 100644 index c097ee41c745c78ae62449baf527ca275f99aed1..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 1500, - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "val_bpb": 1.3280401785061304, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "pre1900-d12-1ep-4sh-r20", - "wandb_run_id": "59d78984", - "wandb_group": "pre1900-d12", - "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", - "tokenizer_fingerprint": "ab10cc8eda78637c", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "pre1900-d12-1ep-4sh-r20", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 4, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "pre1900-d12-1ep-4sh-r20", - "group": "pre1900-d12", - "tags": [ - "pre1900-corpus", - "d12", - "ratio20", - "4shards", - "1epoch" - ] - }, - "config_fingerprint": "27dd1d0b7b2eff44", - "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" - }, - "stage": "base", - "base_experiment_id": "pre1900-d12-1ep-4sh-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "27dd1d0b7b2eff44" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.3280061850201952, - "smooth_train_loss": 3.6858135004402817, - "total_training_time": 3950.216495037079, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_002000.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_002000.json deleted file mode 100644 index 6e45dbda38d25705cb7dc4be3e80bdce10571852..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 2000, - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "val_bpb": 1.2758348491157965, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "pre1900-d12-1ep-4sh-r20", - "wandb_run_id": "59d78984", - "wandb_group": "pre1900-d12", - "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", - "tokenizer_fingerprint": "ab10cc8eda78637c", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "pre1900-d12-1ep-4sh-r20", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 4, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "pre1900-d12-1ep-4sh-r20", - "group": "pre1900-d12", - "tags": [ - "pre1900-corpus", - "d12", - "ratio20", - "4shards", - "1epoch" - ] - }, - "config_fingerprint": "27dd1d0b7b2eff44", - "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" - }, - "stage": "base", - "base_experiment_id": "pre1900-d12-1ep-4sh-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "27dd1d0b7b2eff44" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.2758348491157965, - "smooth_train_loss": 3.5742909025197345, - "total_training_time": 5275.729542255402, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_002500.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_002500.json deleted file mode 100644 index b59c4511fb50eb1d554602ea10ecce8a0d5cc2c4..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_002500.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 2500, - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "val_bpb": 1.2427248605453876, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "pre1900-d12-1ep-4sh-r20", - "wandb_run_id": "59d78984", - "wandb_group": "pre1900-d12", - "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", - "tokenizer_fingerprint": "ab10cc8eda78637c", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "pre1900-d12-1ep-4sh-r20", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 4, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "pre1900-d12-1ep-4sh-r20", - "group": "pre1900-d12", - "tags": [ - "pre1900-corpus", - "d12", - "ratio20", - "4shards", - "1epoch" - ] - }, - "config_fingerprint": "27dd1d0b7b2eff44", - "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" - }, - "stage": "base", - "base_experiment_id": "pre1900-d12-1ep-4sh-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "27dd1d0b7b2eff44" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 13, - "pos": 10792769, - "epoch": 1, - "pq_idx": 13, - "rg_idx": 10792769 - }, - "loop_state": { - "min_val_bpb": 1.2427248605453876, - "smooth_train_loss": 3.3412281067532192, - "total_training_time": 6600.070840597153, - "stage_training_flops": 1162736943759360000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1162736943759360000 - } -} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_003000.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_003000.json deleted file mode 100644 index f0b9b4e88828ab79c3b8e0cd68d73831f0e19a72..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_003000.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 3000, - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "val_bpb": 1.2025249805885556, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "pre1900-d12-1ep-4sh-r20", - "wandb_run_id": "59d78984", - "wandb_group": "pre1900-d12", - "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", - "tokenizer_fingerprint": "ab10cc8eda78637c", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "pre1900-d12-1ep-4sh-r20", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 4, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "pre1900-d12-1ep-4sh-r20", - "group": "pre1900-d12", - "tags": [ - "pre1900-corpus", - "d12", - "ratio20", - "4shards", - "1epoch" - ] - }, - "config_fingerprint": "27dd1d0b7b2eff44", - "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" - }, - "stage": "base", - "base_experiment_id": "pre1900-d12-1ep-4sh-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "27dd1d0b7b2eff44" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 15, - "pos": 72944769, - "epoch": 1, - "pq_idx": 15, - "rg_idx": 72944769 - }, - "loop_state": { - "min_val_bpb": 1.2025249805885556, - "smooth_train_loss": 3.3140674978000466, - "total_training_time": 7926.366828918457, - "stage_training_flops": 1395284332511232000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1395284332511232000 - } -} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_003500.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_003500.json deleted file mode 100644 index 74743aee2c51b33d3380795eb192137e1d77afdd..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_003500.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 3500, - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "val_bpb": 1.1687947775345633, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "pre1900-d12-1ep-4sh-r20", - "wandb_run_id": "59d78984", - "wandb_group": "pre1900-d12", - "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", - "tokenizer_fingerprint": "ab10cc8eda78637c", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "pre1900-d12-1ep-4sh-r20", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 4, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "pre1900-d12-1ep-4sh-r20", - "group": "pre1900-d12", - "tags": [ - "pre1900-corpus", - "d12", - "ratio20", - "4shards", - "1epoch" - ] - }, - "config_fingerprint": "27dd1d0b7b2eff44", - "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" - }, - "stage": "base", - "base_experiment_id": "pre1900-d12-1ep-4sh-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "27dd1d0b7b2eff44" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 18, - "pos": 35096769, - "epoch": 1, - "pq_idx": 18, - "rg_idx": 35096769 - }, - "loop_state": { - "min_val_bpb": 1.1687947775345633, - "smooth_train_loss": 3.0847306468854994, - "total_training_time": 9253.729534626007, - "stage_training_flops": 1627831721263104000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1627831721263104000 - } -} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_004000.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_004000.json deleted file mode 100644 index 236ff6a99fb693d8710570316e441ede01a7209d..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_004000.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 4000, - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "val_bpb": 1.1246669836048595, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "pre1900-d12-1ep-4sh-r20", - "wandb_run_id": "59d78984", - "wandb_group": "pre1900-d12", - "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", - "tokenizer_fingerprint": "ab10cc8eda78637c", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "pre1900-d12-1ep-4sh-r20", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 4, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "pre1900-d12-1ep-4sh-r20", - "group": "pre1900-d12", - "tags": [ - "pre1900-corpus", - "d12", - "ratio20", - "4shards", - "1epoch" - ] - }, - "config_fingerprint": "27dd1d0b7b2eff44", - "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" - }, - "stage": "base", - "base_experiment_id": "pre1900-d12-1ep-4sh-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "27dd1d0b7b2eff44" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 20, - "pos": 97248769, - "epoch": 1, - "pq_idx": 20, - "rg_idx": 97248769 - }, - "loop_state": { - "min_val_bpb": 1.1246669836048595, - "smooth_train_loss": 3.0878954815260107, - "total_training_time": 10581.229209661484, - "stage_training_flops": 1860379110014976000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1860379110014976000 - } -} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_004200.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_004200.json deleted file mode 100644 index 258b03f83162561d7473e8e2a3f668eb5e93e5be..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_004200.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 4200, - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "val_bpb": 1.1155106548699898, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "pre1900-d12-1ep-4sh-r20", - "wandb_run_id": "59d78984", - "wandb_group": "pre1900-d12", - "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", - "tokenizer_fingerprint": "ab10cc8eda78637c", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "pre1900-d12-1ep-4sh-r20", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 4, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "pre1900-d12-1ep-4sh-r20", - "group": "pre1900-d12", - "tags": [ - "pre1900-corpus", - "d12", - "ratio20", - "4shards", - "1epoch" - ] - }, - "config_fingerprint": "27dd1d0b7b2eff44", - "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" - }, - "stage": "base", - "base_experiment_id": "pre1900-d12-1ep-4sh-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "27dd1d0b7b2eff44" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 22, - "pos": 2109569, - "epoch": 1, - "pq_idx": 22, - "rg_idx": 2109569 - }, - "loop_state": { - "min_val_bpb": 1.1155106548699898, - "smooth_train_loss": 3.2450879718089913, - "total_training_time": 11112.257759571075, - "stage_training_flops": 1953398065515724800, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1953398065515724800 - } -} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_000500.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_000500.pt deleted file mode 100644 index af6edd914dc17156af26513c2577cd3d1049a7ba..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:348b87cda555d091e7fb42558bccaa2aa6dc1238dd6292b3826fffe639d3d5ac -size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_001000.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_001000.pt deleted file mode 100644 index bf76cd1fdaaeeaad85c74c46057fc00163b388c5..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:80b2dd0c88e2208837c7784d7453b1b5304cfe9eada1024626c91b945bcbe95b -size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_001500.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_001500.pt deleted file mode 100644 index 08b0222d9ce3d9ee3590cc4d4e6dd7fd4a846082..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:30727a132972164e8f477d30e4258ee98bd747a6747ebc732deebb11eaf973a5 -size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_002000.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_002000.pt deleted file mode 100644 index e72be4ec725a6d2a01f7ddb038e4a8ecc24b7fd9..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f6e98e913a7ce0b076a3a8e246bc0ab0369816071671bbe705287cfd09942f15 -size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_002500.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_002500.pt deleted file mode 100644 index 684fe70424f2685f7efff277adc31090d06fd02b..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0ef0766f989c5016fcb1e3b1125acdf342adc931d6f24984e73b820fe939fb26 -size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_003000.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_003000.pt deleted file mode 100644 index 190e5240a9235222fa8ac11e1088f07fd6926a7e..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_003000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f5227a1eff500535694508ed71c4d9802e7a85f2b562b006aaf289510559cf14 -size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_003500.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_003500.pt deleted file mode 100644 index d804ab27a38111cd4f4d3086285f005d97300a17..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_003500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:06e1588104aeda27382230a02f2c75869ea55a61889d014da02ebbe257988557 -size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_004000.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_004000.pt deleted file mode 100644 index d4dc33785e3f663b644442c576784d732569b54e..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_004000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c6763244f5f3349176b8af99aad687dc155f8ce81ee9c7a111ed6edf356487b6 -size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_004200.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_004200.pt deleted file mode 100644 index 262525b39b3ae4cce7e86f17338503ed64300dfc..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_004200.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:dd0bc3175b836b1306cac76c06f3660e26cf09bacc9f97d4b9dcba226712cda3 -size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_000500_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 697ca7be30689badd4ff00fdea8f29404dcdcd1b..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:de7d1085cac0ffcd273a5edb57c5c3c5eacbb6279b074c43dc687ad6c3c9a145 -size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_001000_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index e69bef4bb415e18568be9d00ed4d2d3756d78e0b..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b9e7a08c9a799a130d1a773eb8b74c987100048eb8155dabe930221aba039b83 -size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_001500_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 98b47d7a3702d17d9407d16b65e36f864a36f809..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:92f4059f97359f2194043da27ed26c8d626af6afeed555fc92ab4a23a15460c4 -size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_002000_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index b0ed4a9b664aff4de3639504ed8cb22245e392d3..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:33214aa57b70f8165ce7fe7fd6d75ec950e01bd58fcf802aa11af5a94ad6bf3a -size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_002500_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_002500_rank0.pt deleted file mode 100644 index add65ddce959c42044cd01849652913d360ff6ad..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8c0a51b3e3d4e809e224e02690c57ec626e3f02d18935a3dcebf5e14f184024a -size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_003000_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_003000_rank0.pt deleted file mode 100644 index 4e963e5e3cf9ce2c144f788c2647301a326a6906..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_003000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e39341560c61a627558edaf0cdefbdaac71db46b193aafdb106ded04199a2d58 -size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_003500_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_003500_rank0.pt deleted file mode 100644 index a29c5235a7349aa44970919af689d9dbb6639aa8..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_003500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:be32cd5d76a6f2d5e4994d20a6d4e912209d9a3765879a1b6664d56931c38486 -size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_004000_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_004000_rank0.pt deleted file mode 100644 index 61833e0fb3127c7b30569ef2d80c7889d27741ae..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_004000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9e873b90b6248044760ba20db49d2310393019b7ab82946cd641274b9ffce449 -size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_004200_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_004200_rank0.pt deleted file mode 100644 index f203f44e44258d8ff7a42e90f2c8160bb5f09cfd..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_004200_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:23eaf1f3bdd7e46887ef59c43633560978293302b4d9326507a87a0b88c8766f -size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/config.json b/experiments/pre1900-d12-1ep-4sh-r20/config.json deleted file mode 100644 index e9b154ac56a36e262ba0803c4c41bc1104f004b4..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/config.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 4, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "pre1900-d12-1ep-4sh-r20", - "group": "pre1900-d12", - "tags": [ - "pre1900-corpus", - "d12", - "ratio20", - "4shards", - "1epoch" - ] - }, - "config_fingerprint": "27dd1d0b7b2eff44", - "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" -} diff --git a/experiments/pre1900-d12-1ep-4sh-r20/evals/core.json b/experiments/pre1900-d12-1ep-4sh-r20/evals/core.json deleted file mode 100644 index 4210a671f0865e8339c49989400f858508f7ba14..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 4200)", - "step": 4200, - "bpb": {}, - "core_metric": 0.08775698798334049, - "core_results": { - "hellaswag_zeroshot": 0.2801234722137451, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.10560503602027893, - "arc_easy": 0.3207070529460907, - "arc_challenge": 0.23037542402744293, - "copa": 0.550000011920929, - "commonsense_qa": 0.32186731696128845, - "piqa": 0.5446137189865112, - "openbook_qa": 0.2460000067949295, - "lambada_openai": 0.3044828176498413, - "hellaswag": 0.2790280878543854, - "winograd": 0.5457875728607178, - "winogrande": 0.5043409466743469, - "bigbench_dyck_languages": 0.12600000202655792, - "agi_eval_lsat_ar": 0.2869565188884735, - "bigbench_cs_algorithms": 0.38712120056152344, - "bigbench_operators": 0.08095238357782364, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.01825922355055809, - "coqa": 0.08142302185297012, - "boolq": 0.6039755344390869, - "bigbench_language_identification": 0.25130000710487366 - }, - "centered_results": { - "hellaswag_zeroshot": 0.04016462961832682, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.10560503602027893, - "arc_easy": 0.0942760705947876, - "arc_challenge": -0.026166101296742756, - "copa": 0.10000002384185791, - "commonsense_qa": 0.15233414620161054, - "piqa": 0.08922743797302246, - "openbook_qa": -0.005333324273427327, - "lambada_openai": 0.3044828176498413, - "hellaswag": 0.038704117139180504, - "winograd": 0.09157514572143555, - "winogrande": 0.008681893348693848, - "bigbench_dyck_languages": 0.12600000202655792, - "agi_eval_lsat_ar": 0.10869564861059187, - "bigbench_cs_algorithms": 0.38712120056152344, - "bigbench_operators": 0.08095238357782364, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.01825922355055809, - "coqa": 0.08142302185297012, - "boolq": -0.04216964621292916, - "bigbench_language_identification": 0.176347642579619 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/evals/samples.json b/experiments/pre1900-d12-1ep-4sh-r20/evals/samples.json deleted file mode 100644 index e91a2b82b4a03f729d0daa919faa8392ad46b143..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 4200)", - "step": 4200, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is now in the hands of the French, and the French in the hands of the" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as that of silver, and the same as that of silver. The" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and then Saturday, and then Sunday, and then Monday, and then" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the hotter, and the hotter the hotter. The hotter is" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: 1. The sun, 2. The moon, 3. The" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a dark brown, with a light shade of green on the back, and a" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is 5, then x is 5, then x is 5, then" - } - ], - "unconditioned_samples": [ - "<|bos|>J. H. Moffatt, Manager.\n\nRAILLE FOR THE MILL.\n\nFor 1892 We're paying to the Post Office the last amount paid into the treasury each year for supplies extras on deposit at this Office:\n\nDtBuAHe FOR THE MILL\n\nfor The last year ,DEARICE post Office Address-which was Michigan State, WIS IN vising - Row of Cheap 75 per cent. cash.\n\nfactice for The last year.\n\nThe problem OF | the tariff Is 10 settle the difference between this country and 1.-THE United States, and this ques- ! tion Is open TO mental complication", - "<|bos|>has an amendment, including the spikes of the bills. The bill, which amends the eighth section, fixes \" ), those which, on motion in the two branches , of the Legislature, must continue to bud, but after the year 1781 shall again refer the ques- in the report of the Joint Select Com mittee on the Territories to select officers in each branch of the House annually provided until their successors are duly elected IN both branches.\n\n2ndIr i:\n\n}. That the Speaker of the House of Representatives, through the Sergeant at Arms, whose duty it is to nominate members OF Congress, shall in all cases select the", - "<|bos|>Again the clock on the chimneypiece strikes the half-hour in the intervals between the half hour and quarter struck by the clock on the house. House. He goes but will return again and tries the house, but the name and the two Cockneys will hate each other\n\nHeit among the neglected and fruitless in all the rambling and leafess sylvanism of a hne beach gossip among a crowd of dilapidated settlers-the eousel Mr. Shields might not turn house nor tree Every loop-hole is broken where battle signals disabled Every door is broken at every point where haste to resume the shuttle or effort to claP people. H", - "<|bos|>inteiful 100, it'll IL sour offi,a;naa pasualizing, form is subdivision of intellect as yo1aah customer and will be both increased and ornamented. The same rule applies 10 al work or source Of strength they are manipulated by the mind When svotper is WORKING IL W h questionable or impure WORK 10 not make specia believe, t94n intinct or working verb, but only 20 assert ability TO make believe 10 disbelieve, and this ability is AL nyISHtPO thoughts afish a,nhnrundian or rom Iast year.\n\nWHEN", - "<|bos|>DWsiuiuiu's LUeIy, with MIR. Grand Tor The defendant.\n\n| Total number OF witnesses by Mrs. | Waters, and she obtained their oath by Walkers | drinking, getting her divorce. | Willagln is only nine i'm bed. A wife is in Jail on charge of lntolerance.\n\nOn Monday the jury returned a verdict OF | murder against John Dora Blair. One OF McPherson's Chickadee Gus 'rais have begun enp!ling Greetings from Ella Benson, the proprietor OF the Pa. | ,ce Hotel.\n\n| To-day, TO the inquiry OF the Bank", - "<|bos|>GRADE\n\nE.W. UXBRIDGE ALPAHAEN\n\nWUXBRIDGE CARCS8) SILASHUEY STELLI AVE\n\nLike unto the fowler with touch of fire,\n\nWhen vivid the lightnings would flash desire:\n\nTheir awakshinc originareth the sky singeth ay the world, with embarrassment abounding aforefhem nameing opportunely to Belvideer Jews Revilejan hght.\n\nElbmergus Borringer.\n\nApppres sahoersativ schonwatter, Erickie frankel marafarsch | Walleton villn", - "<|bos|>rys substances of all percussion caps not ined, or mot has been lowered or made smate, or succeeded to the caloric of oSeH soot-cran, or from some other chief substance; otherwise it viii be temper-\n\nthe heat will not make Perseus little unless virtue and Wealth require it too, barely say newspaper paper! There be many lovers IN tilins But let them bear in mind the\n\nfirst principles in the selectione Of speeches, and in the popular orator, and rather in\n\nthe popular tale than the truth itself; so that where it\n\ndoes occur, as in familiar instances,", - "<|bos|>The figurative language would be appropriate in a town of town ships, but it is one of the affairs of the school to speak of the town home in the home, and is then one of the suggestions for those who are trying to find out how to understand the school term.\n\nThe figurative expressions would be jesuitous and those allusions purely aphoristic. It is only as much of the styles in play as hasn't the life really in it that it then bears. Unless the figurative language has a genuine article it is unattainable. where we have only the elastic spirit, but as much" - ] -} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/evals/val_bpb.json b/experiments/pre1900-d12-1ep-4sh-r20/evals/val_bpb.json deleted file mode 100644 index 58b9124964272144bc9ff2e0bd2012405bfdbb00..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 4200)", - "step": 4200, - "bpb": { - "val": 1.080046782192013 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/run.json b/experiments/pre1900-d12-1ep-4sh-r20/run.json deleted file mode 100644 index d4a8373fa87093fb4984f6a2bdb6ecd48251cef1..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "stage": "base", - "base_experiment_id": "pre1900-d12-1ep-4sh-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "27dd1d0b7b2eff44", - "wandb_run_id": "59d78984", - "created_at": 1781914327 -} diff --git a/experiments/pre1900-d12-1ep-4sh-r20/summary.json b/experiments/pre1900-d12-1ep-4sh-r20/summary.json deleted file mode 100644 index 491cd7bafc30f4d2c821fa0d30d080d2bf8120d2..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "stage": "base", - "base_experiment_id": "pre1900-d12-1ep-4sh-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "mhla/pre1900-corpus", - "dataset_revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "step": 4200, - "depth": 12, - "target_param_data_ratio": 20.0, - "training_tokens": 2202009600, - "final_sampled_val_bpb": 1.1155106548699898, - "minimum_sampled_val_bpb": 1.1155106548699898, - "full_val_bpb": 1.080046782192013, - "core_metric": 0.08775698798334049, - "centered_results": { - "hellaswag_zeroshot": 0.04016462961832682, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.10560503602027893, - "arc_easy": 0.0942760705947876, - "arc_challenge": -0.026166101296742756, - "copa": 0.10000002384185791, - "commonsense_qa": 0.15233414620161054, - "piqa": 0.08922743797302246, - "openbook_qa": -0.005333324273427327, - "lambada_openai": 0.3044828176498413, - "hellaswag": 0.038704117139180504, - "winograd": 0.09157514572143555, - "winogrande": 0.008681893348693848, - "bigbench_dyck_languages": 0.12600000202655792, - "agi_eval_lsat_ar": 0.10869564861059187, - "bigbench_cs_algorithms": 0.38712120056152344, - "bigbench_operators": 0.08095238357782364, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.01825922355055809, - "coqa": 0.08142302185297012, - "boolq": -0.04216964621292916, - "bigbench_language_identification": 0.176347642579619 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is now in the hands of the French, and the French in the hands of the" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as that of silver, and the same as that of silver. The" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and then Saturday, and then Sunday, and then Monday, and then" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the hotter, and the hotter the hotter. The hotter is" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: 1. The sun, 2. The moon, 3. The" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a dark brown, with a light shade of green on the back, and a" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is 5, then x is 5, then x is 5, then" - } - ], - "unconditioned_samples": [ - "<|bos|>J. H. Moffatt, Manager.\n\nRAILLE FOR THE MILL.\n\nFor 1892 We're paying to the Post Office the last amount paid into the treasury each year for supplies extras on deposit at this Office:\n\nDtBuAHe FOR THE MILL\n\nfor The last year ,DEARICE post Office Address-which was Michigan State, WIS IN vising - Row of Cheap 75 per cent. cash.\n\nfactice for The last year.\n\nThe problem OF | the tariff Is 10 settle the difference between this country and 1.-THE United States, and this ques- ! tion Is open TO mental complication", - "<|bos|>has an amendment, including the spikes of the bills. The bill, which amends the eighth section, fixes \" ), those which, on motion in the two branches , of the Legislature, must continue to bud, but after the year 1781 shall again refer the ques- in the report of the Joint Select Com mittee on the Territories to select officers in each branch of the House annually provided until their successors are duly elected IN both branches.\n\n2ndIr i:\n\n}. That the Speaker of the House of Representatives, through the Sergeant at Arms, whose duty it is to nominate members OF Congress, shall in all cases select the", - "<|bos|>Again the clock on the chimneypiece strikes the half-hour in the intervals between the half hour and quarter struck by the clock on the house. House. He goes but will return again and tries the house, but the name and the two Cockneys will hate each other\n\nHeit among the neglected and fruitless in all the rambling and leafess sylvanism of a hne beach gossip among a crowd of dilapidated settlers-the eousel Mr. Shields might not turn house nor tree Every loop-hole is broken where battle signals disabled Every door is broken at every point where haste to resume the shuttle or effort to claP people. H", - "<|bos|>inteiful 100, it'll IL sour offi,a;naa pasualizing, form is subdivision of intellect as yo1aah customer and will be both increased and ornamented. The same rule applies 10 al work or source Of strength they are manipulated by the mind When svotper is WORKING IL W h questionable or impure WORK 10 not make specia believe, t94n intinct or working verb, but only 20 assert ability TO make believe 10 disbelieve, and this ability is AL nyISHtPO thoughts afish a,nhnrundian or rom Iast year.\n\nWHEN", - "<|bos|>DWsiuiuiu's LUeIy, with MIR. Grand Tor The defendant.\n\n| Total number OF witnesses by Mrs. | Waters, and she obtained their oath by Walkers | drinking, getting her divorce. | Willagln is only nine i'm bed. A wife is in Jail on charge of lntolerance.\n\nOn Monday the jury returned a verdict OF | murder against John Dora Blair. One OF McPherson's Chickadee Gus 'rais have begun enp!ling Greetings from Ella Benson, the proprietor OF the Pa. | ,ce Hotel.\n\n| To-day, TO the inquiry OF the Bank", - "<|bos|>GRADE\n\nE.W. UXBRIDGE ALPAHAEN\n\nWUXBRIDGE CARCS8) SILASHUEY STELLI AVE\n\nLike unto the fowler with touch of fire,\n\nWhen vivid the lightnings would flash desire:\n\nTheir awakshinc originareth the sky singeth ay the world, with embarrassment abounding aforefhem nameing opportunely to Belvideer Jews Revilejan hght.\n\nElbmergus Borringer.\n\nApppres sahoersativ schonwatter, Erickie frankel marafarsch | Walleton villn", - "<|bos|>rys substances of all percussion caps not ined, or mot has been lowered or made smate, or succeeded to the caloric of oSeH soot-cran, or from some other chief substance; otherwise it viii be temper-\n\nthe heat will not make Perseus little unless virtue and Wealth require it too, barely say newspaper paper! There be many lovers IN tilins But let them bear in mind the\n\nfirst principles in the selectione Of speeches, and in the popular orator, and rather in\n\nthe popular tale than the truth itself; so that where it\n\ndoes occur, as in familiar instances,", - "<|bos|>The figurative language would be appropriate in a town of town ships, but it is one of the affairs of the school to speak of the town home in the home, and is then one of the suggestions for those who are trying to find out how to understand the school term.\n\nThe figurative expressions would be jesuitous and those allusions purely aphoristic. It is only as much of the styles in play as hasn't the life really in it that it then bears. Unless the figurative language has a genuine article it is unattainable. where we have only the elastic spirit, but as much" - ], - "training_time_seconds": 11112.257759571075, - "stage_training_flops": 1.9533980655157248e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.9533980655157248e+18, - "config_fingerprint": "27dd1d0b7b2eff44", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/59d78984", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/pre1900-d12-1ep-4sh-r20", - "dataset_fingerprint": "b99d24e50808d3c5", - "tokenizer_fingerprint": "ab10cc8eda78637c", - "unique_train_tokens": 2268069888, - "effective_epochs": 0.970873786407767 -} diff --git a/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/experiment_tokenizer.json b/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 5efd600c65d36b94b1fca1a6e5de698b995d6882..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "pre1900-d12-1ep-4sh-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 4, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1781914360 -} diff --git a/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/token_bytes.pt b/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/token_bytes.pt deleted file mode 100644 index cc2638a46b3c5e02d8e316d6ceb532db60e5da83..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:98cdf539455a9dd9b63f5f4303da19912fb91d49dec505a80f01c4e13e60c38c -size 132649 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/tokenizer.pkl b/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/tokenizer.pkl deleted file mode 100644 index 567de529d282e5eda527aee448dbd68ca63f60f9..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d4d66008f3953c4c06658f288439a51cc27967d15e0ffa730e154f4de3c3109e -size 398115 diff --git a/experiments/pre1900-d12-1ep-r20/config.json b/experiments/pre1900-d12-1ep-r20/config.json deleted file mode 100644 index 03302b08d83c9ec554f0b8b3e63164c82c75185d..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-r20/config.json +++ /dev/null @@ -1,74 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "pre1900-d12-1ep-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 41, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 2268069888, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "pre1900-d12-1ep-r20", - "group": "pre1900-d12", - "tags": [ - "pre1900", - "pre1900-corpus", - "d12", - "ratio20", - "r20", - "a100", - "bf16" - ] - }, - "config_fingerprint": "5faa20d56558c829", - "artifact_path": "experiments/pre1900-d12-1ep-r20" -} diff --git a/experiments/pre1900-d12-1ep-r20/run.json b/experiments/pre1900-d12-1ep-r20/run.json deleted file mode 100644 index bebf175cdd289fe98ebb9796ceff727d6956ef90..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-r20/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "pre1900-d12-1ep-r20", - "stage": "base", - "base_experiment_id": "pre1900-d12-1ep-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "5faa20d56558c829", - "wandb_run_id": "da8d0a59", - "created_at": 1781898389 -} diff --git a/experiments/pre1900-d12-1ep-r20/tokenizer/experiment_tokenizer.json b/experiments/pre1900-d12-1ep-r20/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 5284c308a121251d0c0a3b9f4a1bb23de95fec68..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-r20/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "pre1900-d12-1ep-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "mhla/pre1900-corpus", - "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", - "validation_shard": 41, - "num_train_shards": 41, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1781898707 -} diff --git a/experiments/pre1900-d12-1ep-r20/tokenizer/token_bytes.pt b/experiments/pre1900-d12-1ep-r20/tokenizer/token_bytes.pt deleted file mode 100644 index af63f54ce793582c533bb911725940394e3fec49..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-r20/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:93c2b481ec04debdcd92e3ad57b8ee5c0bfe806f6e85fc9aa1cb3111de4e6d84 -size 132649 diff --git a/experiments/pre1900-d12-1ep-r20/tokenizer/tokenizer.pkl b/experiments/pre1900-d12-1ep-r20/tokenizer/tokenizer.pkl deleted file mode 100644 index e78e2463ea2d42fe9eed62f1948f45ace1d6faf8..0000000000000000000000000000000000000000 --- a/experiments/pre1900-d12-1ep-r20/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3c17e58522a9fd24956de85e56cd23d5f2131bb4a0d9af3433647fb8942add5b -size 397922 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_000500.json b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_000500.json deleted file mode 100644 index 21925f13a4f63a1562871eba5b8138cad0cd7521..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "val_bpb": 1.3216810908332168, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-25sh-r11-wd42", - "wandb_run_id": "4e526b5a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11,25shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/config.json", - "tokenizer_fingerprint": "ebb3705d7792a34d", - "git_commit_sha": "7e29503cca7b67e1418323c3628e6f935e4feacf", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-25sh-r11-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-25sh-r11-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11", - "25shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "47bfa49108766b7d", - "artifact_path": "experiments/think-d12-1ep-25sh-r11-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-25sh-r11-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "47bfa49108766b7d" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3216810908332168, - "smooth_train_loss": 3.7465076848716428, - "total_training_time": 1308.3161821365356, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_001000.json b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_001000.json deleted file mode 100644 index 1dc913105ff148b606108ba7d44ee6c225f5c541..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "val_bpb": 1.2426550595586172, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-25sh-r11-wd42", - "wandb_run_id": "4e526b5a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11,25shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/config.json", - "tokenizer_fingerprint": "ebb3705d7792a34d", - "git_commit_sha": "7e29503cca7b67e1418323c3628e6f935e4feacf", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-25sh-r11-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-25sh-r11-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11", - "25shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "47bfa49108766b7d", - "artifact_path": "experiments/think-d12-1ep-25sh-r11-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-25sh-r11-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "47bfa49108766b7d" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2426550595586172, - "smooth_train_loss": 3.5992329357223416, - "total_training_time": 2646.5828285217285, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_001500.json b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_001500.json deleted file mode 100644 index c41eed852092dd0dd83f743c0b2dfd677c12997f..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "val_bpb": 1.1829778736744743, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-25sh-r11-wd42", - "wandb_run_id": "4e526b5a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11,25shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/config.json", - "tokenizer_fingerprint": "ebb3705d7792a34d", - "git_commit_sha": "7e29503cca7b67e1418323c3628e6f935e4feacf", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-25sh-r11-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-25sh-r11-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11", - "25shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "47bfa49108766b7d", - "artifact_path": "experiments/think-d12-1ep-25sh-r11-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-25sh-r11-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "47bfa49108766b7d" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.1829778736744743, - "smooth_train_loss": 3.4545608734819186, - "total_training_time": 3986.5272040367126, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_002000.json b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_002000.json deleted file mode 100644 index 0b829f3e6d4d0b956cf57e1b09a1873fd2b03447..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "val_bpb": 1.1274373944966156, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-25sh-r11-wd42", - "wandb_run_id": "4e526b5a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11,25shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/config.json", - "tokenizer_fingerprint": "ebb3705d7792a34d", - "git_commit_sha": "7e29503cca7b67e1418323c3628e6f935e4feacf", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-25sh-r11-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-25sh-r11-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11", - "25shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "47bfa49108766b7d", - "artifact_path": "experiments/think-d12-1ep-25sh-r11-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-25sh-r11-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "47bfa49108766b7d" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.1274373944966156, - "smooth_train_loss": 3.19770985990571, - "total_training_time": 5324.934033155441, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_002362.json b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_002362.json deleted file mode 100644 index f63af5b1eb3294f4bbd7551c278ba6885908febe..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,142 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "val_bpb": 1.103670265641304, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-25sh-r11-wd42", - "wandb_run_id": "4e526b5a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11,25shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/config.json", - "tokenizer_fingerprint": "ebb3705d7792a34d", - "git_commit_sha": "7e29503cca7b67e1418323c3628e6f935e4feacf", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-25sh-r11-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-25sh-r11-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11", - "25shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "47bfa49108766b7d", - "artifact_path": "experiments/think-d12-1ep-25sh-r11-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-25sh-r11-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "47bfa49108766b7d" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": 1.103670265641304, - "smooth_train_loss": 3.1383014196569707, - "total_training_time": 6293.967695713043, - "stage_training_flops": 1098553864463843328, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1098553864463843328 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_000500.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_000500.pt deleted file mode 100644 index 23f325a09c26660d57660a524dd13516c47bb557..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2ba7f29f5abc6b63320f4acbaf91d9bfdb0a6cfb5ec5214e22fd4b4e7cacde56 -size 792761690 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_001000.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_001000.pt deleted file mode 100644 index e51cf0f090eec510c66d82c0eee4a292bbf950df..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2cf5c5d3bab8b0a98284389f71e40abef1a253000e05ba9c9bac41931767af40 -size 792761690 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_001500.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_001500.pt deleted file mode 100644 index c4b02f48e6ddb7ac6fdb778a5fc2408097150054..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:47130e6261cf39ddc15a667483465368a1ede8fd96d4d4b1c75c4bf3deb1a8cc -size 792761690 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_002000.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_002000.pt deleted file mode 100644 index 50164f1c86395a7295ec8feb0306ffb8dda49456..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e1c858d0142791d42fcab3b5fbbd72d188032bc399f33da736ddf1ef864d5c0e -size 792761690 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_002362.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_002362.pt deleted file mode 100644 index df8a17957e7c45b557bd477ccd73d02694d6b737..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8b2299fe338c35e226e044e2f3555c7ec5cc6fd8d59317db2235590d01aef9d1 -size 792761690 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index f1f1ca6963076c7bf1d346af21583cd2e0e273a7..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f37e7f554852ced4488a09521f28a55c40c1708feb4a969c82444cc52866cb9d -size 1246165357 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index b53953acb1eb6a9440d07528f9f0860efc8f86e9..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:45b9c62b9529b4e3da87ef073be54ea34562194f9810af830675fadb2d2dc2d1 -size 1246165357 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 52ae32f37959deed46455181bdd09f00ec1cb3fb..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9d18f0823e9354165fbbe28dff83bd45cd3cface494ae994a5465340f8c23109 -size 1246165357 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 59e834a6b7188d20c84621317487f11a001a71ad..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:13abe8ff459d4a01082c70ed29497a8af2747cc3e94c7c298c3e83ff568048b2 -size 1246165357 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_002362_rank0.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 17bdc04253494be5723658c443c83c965a7002de..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8607ee156d616769b388c071de7dd3bc068f04fb8196f81843357ddb59c71dc2 -size 1246165357 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/config.json b/experiments/think-d12-1ep-25sh-r11-wd42/config.json deleted file mode 100644 index 6dcbeacb7abdcf69a3092a24acfe5f8092defa70..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/config.json +++ /dev/null @@ -1,59 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-25sh-r11-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11", - "25shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "47bfa49108766b7d", - "artifact_path": "experiments/think-d12-1ep-25sh-r11-wd42" -} diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/evals/val_bpb.json b/experiments/think-d12-1ep-25sh-r11-wd42/evals/val_bpb.json deleted file mode 100644 index 9611aa080f82197f0797867dbca4c42b99fb733d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.0526348691238439 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/run.json b/experiments/think-d12-1ep-25sh-r11-wd42/run.json deleted file mode 100644 index 5263e8b5c526a64212bdbfd68eca0b704ea98092..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "stage": "base", - "base_experiment_id": "think-d12-1ep-25sh-r11-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "47bfa49108766b7d", - "wandb_run_id": "4e526b5a", - "created_at": 1782495190 -} diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/summary.json b/experiments/think-d12-1ep-25sh-r11-wd42/summary.json deleted file mode 100644 index 2175bed11026365f7d85223d427e8fdc4f5fd599..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/summary.json +++ /dev/null @@ -1,32 +0,0 @@ -{ - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "stage": "base", - "base_experiment_id": "think-d12-1ep-25sh-r11-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.103670265641304, - "minimum_sampled_val_bpb": 1.103670265641304, - "full_val_bpb": 1.0526348691238439, - "core_metric": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [], - "training_time_seconds": 6293.967695713043, - "stage_training_flops": 1.0985538644638433e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.0985538644638433e+18, - "config_fingerprint": "47bfa49108766b7d", - "git_commit_sha": "7e29503cca7b67e1418323c3628e6f935e4feacf", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/4e526b5a", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-1ep-25sh-r11-wd42", - "dataset_fingerprint": "a6e1b3a100e0d8b3", - "tokenizer_fingerprint": "ebb3705d7792a34d", - "unique_train_tokens": 1275519304, - "effective_epochs": 0.970873786164196 -} diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer/experiment_tokenizer.json b/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 01564e84d532ee9e205955aaffee1d865c4b28f0..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "think-d12-1ep-25sh-r11-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1782495209 -} diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer/token_bytes.pt b/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer/token_bytes.pt deleted file mode 100644 index 80dbb386d071538021ab399ee6e965ae0cd1a54e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:59f928e04aa2ac37dd4064493240d1e73ecab7acb217c5a183311b0c523a3468 -size 132649 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer/tokenizer.pkl b/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer/tokenizer.pkl deleted file mode 100644 index a812730c3de1acc1e9e30ef6c4ccf1e9360f5d8a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fd583e3c35851c62295a1a0f6d688923e4f30649ac963b701ec2440fe8bc3e4f -size 404221 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_000500.json b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_000500.json deleted file mode 100644 index 59f658d8c9d9655b7ddf9466f4304ecc0d20e9d4..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 500, - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "val_bpb": 1.3221519749129604, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-44sh-r20-wd42", - "wandb_run_id": "5c4fba8a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,44shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/config.json", - "tokenizer_fingerprint": "1744d7b7ee0d5d80", - "git_commit_sha": "fd4d149a650d0fb07ae9b96864bb40e5769622d0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-44sh-r20-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-44sh-r20-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "44shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "a8d632854c2cd1bd", - "artifact_path": "experiments/think-d12-1ep-44sh-r20-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-44sh-r20-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a8d632854c2cd1bd" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3221519749129604, - "smooth_train_loss": 3.7478778179789605, - "total_training_time": 1305.7518684864044, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_001000.json b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_001000.json deleted file mode 100644 index 7ad198c79884fcd6bfa13ebc8fb085602b7b4960..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 1000, - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "val_bpb": 1.2600323113203347, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-44sh-r20-wd42", - "wandb_run_id": "5c4fba8a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,44shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/config.json", - "tokenizer_fingerprint": "1744d7b7ee0d5d80", - "git_commit_sha": "fd4d149a650d0fb07ae9b96864bb40e5769622d0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-44sh-r20-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-44sh-r20-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "44shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "a8d632854c2cd1bd", - "artifact_path": "experiments/think-d12-1ep-44sh-r20-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-44sh-r20-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a8d632854c2cd1bd" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2600323113203347, - "smooth_train_loss": 3.646794584039554, - "total_training_time": 2640.877459049225, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_001500.json b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_001500.json deleted file mode 100644 index 9f43a27678e300b6474f88abba476e3cae89d80a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 1500, - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "val_bpb": 1.2377503150891866, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-44sh-r20-wd42", - "wandb_run_id": "5c4fba8a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,44shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/config.json", - "tokenizer_fingerprint": "1744d7b7ee0d5d80", - "git_commit_sha": "fd4d149a650d0fb07ae9b96864bb40e5769622d0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-44sh-r20-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-44sh-r20-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "44shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "a8d632854c2cd1bd", - "artifact_path": "experiments/think-d12-1ep-44sh-r20-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-44sh-r20-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a8d632854c2cd1bd" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.2377503150891866, - "smooth_train_loss": 3.6301349812757526, - "total_training_time": 3977.6890711784363, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_002000.json b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_002000.json deleted file mode 100644 index 16044dcb616400cb3c75f4feeed101b1751a8069..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 2000, - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "val_bpb": 1.2014593386689274, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-44sh-r20-wd42", - "wandb_run_id": "5c4fba8a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,44shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/config.json", - "tokenizer_fingerprint": "1744d7b7ee0d5d80", - "git_commit_sha": "fd4d149a650d0fb07ae9b96864bb40e5769622d0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-44sh-r20-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-44sh-r20-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "44shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "a8d632854c2cd1bd", - "artifact_path": "experiments/think-d12-1ep-44sh-r20-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-44sh-r20-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a8d632854c2cd1bd" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.2014593386689274, - "smooth_train_loss": 3.4382217869051193, - "total_training_time": 5322.514421463013, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_002500.json b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_002500.json deleted file mode 100644 index 4c7afc7891e2611d2ce054d188a3d16ed39f14ae..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_002500.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 2500, - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "val_bpb": 1.166274403385848, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-44sh-r20-wd42", - "wandb_run_id": "5c4fba8a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,44shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/config.json", - "tokenizer_fingerprint": "1744d7b7ee0d5d80", - "git_commit_sha": "fd4d149a650d0fb07ae9b96864bb40e5769622d0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-44sh-r20-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-44sh-r20-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "44shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "a8d632854c2cd1bd", - "artifact_path": "experiments/think-d12-1ep-44sh-r20-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-44sh-r20-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a8d632854c2cd1bd" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 13, - "pos": 10792769, - "epoch": 1, - "pq_idx": 13, - "rg_idx": 10792769 - }, - "loop_state": { - "min_val_bpb": 1.166274403385848, - "smooth_train_loss": 3.265478801787732, - "total_training_time": 6658.766751766205, - "stage_training_flops": 1162736943759360000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1162736943759360000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_003000.json b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_003000.json deleted file mode 100644 index c8041b4c09d2307652cf533a20e28d08de809019..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_003000.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 3000, - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "val_bpb": 1.1420328141160099, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-44sh-r20-wd42", - "wandb_run_id": "5c4fba8a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,44shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/config.json", - "tokenizer_fingerprint": "1744d7b7ee0d5d80", - "git_commit_sha": "fd4d149a650d0fb07ae9b96864bb40e5769622d0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-44sh-r20-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-44sh-r20-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "44shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "a8d632854c2cd1bd", - "artifact_path": "experiments/think-d12-1ep-44sh-r20-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-44sh-r20-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a8d632854c2cd1bd" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 15, - "pos": 72944769, - "epoch": 1, - "pq_idx": 15, - "rg_idx": 72944769 - }, - "loop_state": { - "min_val_bpb": 1.1420328141160099, - "smooth_train_loss": 3.1094901625575497, - "total_training_time": 8003.383926391602, - "stage_training_flops": 1395284332511232000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1395284332511232000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_003500.json b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_003500.json deleted file mode 100644 index c54a62105182496220f719255a66ed5fcd4ac91d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_003500.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 3500, - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "val_bpb": 1.1108293726096388, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-44sh-r20-wd42", - "wandb_run_id": "5c4fba8a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,44shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/config.json", - "tokenizer_fingerprint": "1744d7b7ee0d5d80", - "git_commit_sha": "fd4d149a650d0fb07ae9b96864bb40e5769622d0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-44sh-r20-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-44sh-r20-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "44shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "a8d632854c2cd1bd", - "artifact_path": "experiments/think-d12-1ep-44sh-r20-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-44sh-r20-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a8d632854c2cd1bd" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 18, - "pos": 35096769, - "epoch": 1, - "pq_idx": 18, - "rg_idx": 35096769 - }, - "loop_state": { - "min_val_bpb": 1.1108293726096388, - "smooth_train_loss": 3.0718457586789576, - "total_training_time": 9347.049030542374, - "stage_training_flops": 1627831721263104000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1627831721263104000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_004000.json b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_004000.json deleted file mode 100644 index 8e7721ac608c77baaa5ec9bbe2c224e7ab18f6f8..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_004000.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 4000, - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "val_bpb": 1.0845218539469001, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-44sh-r20-wd42", - "wandb_run_id": "5c4fba8a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,44shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/config.json", - "tokenizer_fingerprint": "1744d7b7ee0d5d80", - "git_commit_sha": "fd4d149a650d0fb07ae9b96864bb40e5769622d0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-44sh-r20-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-44sh-r20-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "44shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "a8d632854c2cd1bd", - "artifact_path": "experiments/think-d12-1ep-44sh-r20-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-44sh-r20-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a8d632854c2cd1bd" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 20, - "pos": 97248769, - "epoch": 1, - "pq_idx": 20, - "rg_idx": 97248769 - }, - "loop_state": { - "min_val_bpb": 1.0845218539469001, - "smooth_train_loss": 2.94610128781119, - "total_training_time": 10685.518072605133, - "stage_training_flops": 1860379110014976000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1860379110014976000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_004200.json b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_004200.json deleted file mode 100644 index 30c2ced2fd8737ce78266ad95b8535f0fc11d037..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/meta_004200.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 4200, - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "val_bpb": 1.078128321044417, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-44sh-r20-wd42", - "wandb_run_id": "5c4fba8a", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,44shards,1epoch,wd.42", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.42, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-44sh-r20-wd42/config.json", - "tokenizer_fingerprint": "1744d7b7ee0d5d80", - "git_commit_sha": "fd4d149a650d0fb07ae9b96864bb40e5769622d0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-44sh-r20-wd42", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-44sh-r20-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "44shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "a8d632854c2cd1bd", - "artifact_path": "experiments/think-d12-1ep-44sh-r20-wd42" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-44sh-r20-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a8d632854c2cd1bd" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 22, - "pos": 2109569, - "epoch": 1, - "pq_idx": 22, - "rg_idx": 2109569 - }, - "loop_state": { - "min_val_bpb": 1.078128321044417, - "smooth_train_loss": 2.8475676426206853, - "total_training_time": 11219.814347743988, - "stage_training_flops": 1953398065515724800, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1953398065515724800 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_000500.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_000500.pt deleted file mode 100644 index 169cfe116af1b113e56004f1e61f4151ae75537f..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:db546864ca01fbe6875d9c12a049bf0aa4232ba3a4dbdba03ecbdd1095a15e55 -size 792761690 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_001000.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_001000.pt deleted file mode 100644 index 8970bffdd59cd1eeb8f8c0e92b26bbb795ef87ba..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5af8aecdefdbf085e32749ea53e8ce55622196b42c90634282458c34af4d5e1a -size 792761690 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_001500.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_001500.pt deleted file mode 100644 index 7cf46a2b216866d8eb3279f23c0f691b16fb313d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b7582f0335e8a47d987ffb85174e35b14fdb939ea858448b32e21c56d3384056 -size 792761690 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_002000.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_002000.pt deleted file mode 100644 index 66fd446a907daec718a8277446503de47ab352c3..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c8ead5070dd054e0ea8f29381274296936bedcbc712ffd4fd64b11adbda9e088 -size 792761690 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_002500.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_002500.pt deleted file mode 100644 index b98f0b247fe4a9b1ee650d370525abb920e51265..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:77f5679e681a78bf061e9cf261a005a4c658b107e974be777f85e489d8ad1140 -size 792761690 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_003000.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_003000.pt deleted file mode 100644 index f2378febb625e9cb71754ab214236320def49fe4..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_003000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6ca8be32656940ecf969903f3b513a3e673c74e5e405322c584f57e85a13df0f -size 792761690 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_003500.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_003500.pt deleted file mode 100644 index ffa2121a48300e02c9db6372a5b746c52fd89c9c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_003500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b330320b63a95624ecc9ba05722dda9609593987c48a6ab90047c94c9f3d06f1 -size 792761690 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_004000.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_004000.pt deleted file mode 100644 index 32073807e5fa8af036bf75f0ed2a635bf315de37..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_004000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b7986875a99712f761a3f218dfefbca73547a1ae8b45f08ab9d95c7dec1b2a3f -size 792761690 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_004200.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_004200.pt deleted file mode 100644 index cec3bdf5fb68d41c36445e868ca6d34f05b4b472..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/model_004200.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d641bed4b7bf0e787b469394a0c01c73d1d66655be72525a4c625e3f04c961a6 -size 792761690 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 1c7073c47e64bfcaa6f5da191c51485a5d5ab7a5..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1525cb06d3f15e13db9115c5bb67d4ba9c751c6c2bc2a5901d615a785323bd3f -size 1246165357 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index e563a9b78302e96c954592f625012067e807c95a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1181d48ea007611efdd9c45275aa6cbc350d176bf736aef963b05953f4e2837f -size 1246165357 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 8d78655938b9a98d8e34e6e6c9ba10e889f82fcc..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:93fcab3d3b289e92644d578ebcc8b9ea9dd644fe3e36efd4ebda802e0f438b42 -size 1246165357 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index ec5083fdf7f2c48abf607333b9a1cad50e925df1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6602307370d1412fd6f58d5e1b67da2edd22590385d684ec62d1640075409d99 -size 1246165357 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_002500_rank0.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_002500_rank0.pt deleted file mode 100644 index 1993a11840496c591652fcd3e05a7eef136477c7..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fde77b3022a5126a82a110fb379c4aec5775aa6440623ea42a6790c99a721422 -size 1246165357 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_003000_rank0.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_003000_rank0.pt deleted file mode 100644 index 17da371afdca0ab56fa2c3f7051e3bc1bd87cb2c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_003000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c081ca79e912393a33cf8e63d1795047a87d90735757f51df8ed38477ba851d5 -size 1246165357 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_003500_rank0.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_003500_rank0.pt deleted file mode 100644 index 8580d78e2d37cf857ed203cb500371b6bef7df72..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_003500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:328e0b85a4407f35d10e644574bad923147a25c24dfbc5e8ebf0b095c2424497 -size 1246165357 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_004000_rank0.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_004000_rank0.pt deleted file mode 100644 index b17f70cc46f019e2eb03de9327732f713f34901e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_004000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:704c4809a117c46020fcc551cfef13bd6aa5259a74ce8ab36acfd9993b126436 -size 1246165357 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_004200_rank0.pt b/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_004200_rank0.pt deleted file mode 100644 index fa3c696227f942e24fcde49d4fe324c97b7dac97..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/base_checkpoints/optim_004200_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8c8ff3203701457409f4a8b416d178fb66b3ced9764bda87bccf48e0c1990cb6 -size 1246165357 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/config.json b/experiments/think-d12-1ep-44sh-r20-wd42/config.json deleted file mode 100644 index e87fa0fbd8c1e8dbf89b141b813891483f72d194..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/config.json +++ /dev/null @@ -1,59 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "weight_decay": 0.42, - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-44sh-r20-wd42", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "44shards", - "1epoch", - "wd.42" - ] - }, - "config_fingerprint": "a8d632854c2cd1bd", - "artifact_path": "experiments/think-d12-1ep-44sh-r20-wd42" -} diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/evals/core.json b/experiments/think-d12-1ep-44sh-r20-wd42/evals/core.json deleted file mode 100644 index baa7f31846ea732ba51513ea0abd5f9215d4b752..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 4200)", - "step": 4200, - "bpb": {}, - "core_metric": 0.07908813484502109, - "core_results": { - "hellaswag_zeroshot": 0.28251343965530396, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.07814575731754303, - "arc_easy": 0.31607744097709656, - "arc_challenge": 0.2022184282541275, - "copa": 0.5699999928474426, - "commonsense_qa": 0.312039315700531, - "piqa": 0.5527747273445129, - "openbook_qa": 0.24800001084804535, - "lambada_openai": 0.26043081283569336, - "hellaswag": 0.2815176248550415, - "winograd": 0.5604395866394043, - "winogrande": 0.4980268180370331, - "bigbench_dyck_languages": 0.11500000208616257, - "agi_eval_lsat_ar": 0.260869562625885, - "bigbench_cs_algorithms": 0.4015151262283325, - "bigbench_operators": 0.10476190596818924, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.025922421365976334, - "coqa": 0.0899411216378212, - "boolq": 0.542201817035675, - "bigbench_language_identification": 0.25669997930526733 - }, - "centered_results": { - "hellaswag_zeroshot": 0.04335125287373861, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.07814575731754303, - "arc_easy": 0.08810325463612874, - "arc_challenge": -0.06370876232783, - "copa": 0.13999998569488525, - "commonsense_qa": 0.14004914462566373, - "piqa": 0.10554945468902588, - "openbook_qa": -0.002666652202606201, - "lambada_openai": 0.26043081283569336, - "hellaswag": 0.04202349980672201, - "winograd": 0.1208791732788086, - "winogrande": -0.003946363925933838, - "bigbench_dyck_languages": 0.11500000208616257, - "agi_eval_lsat_ar": 0.07608695328235625, - "bigbench_cs_algorithms": 0.4015151262283325, - "bigbench_operators": 0.10476190596818924, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.025922421365976334, - "coqa": 0.0899411216378212, - "boolq": -0.20473206043243405, - "bigbench_language_identification": 0.1822882060563997 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/evals/samples.json b/experiments/think-d12-1ep-44sh-r20-wd42/evals/samples.json deleted file mode 100644 index 24c18830774960bc64d77adec7f2ffeaa2c81052..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 4200)", - "step": 4200, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the world. \n\nThe capital of the world is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the gold of the world. The gold of the world is the" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be the last day of the week. \n\nI am, dear Sir, your most" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the best of all. \n\nThe best of all is the best of all." - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, which is the sun of the solar system. \n\n" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the same as that of the sun, and the same as that of the moon" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the first and second, and x is the number of the second" - } - ], - "unconditioned_samples": [ - "<|bos|>IT\n\nS. PERMISSION Dedicated to the CSTRORRS OF BOSTON. \n\nADVERTISEMENT \n\nWHILE I am following up the translation of the \"Saul,\" it will not be useless to point out to the different subscribers, and to each subscriber, the different phases, phases and phases of the phenomena presented - an appearance presented in all its phases and genera, and common to them all, so as to convey to all minds then present, an impression which is correct as it is true. \n\nYet a work is what it professes to be, and this, to a very considerable extent, ought to go on for ever", - "<|bos|>900 women, or 96,000 or \n\n105,000 inhabitants, and in general the Morningags. Their occupation was not pleasant. . Some of them were married, and had children. Of these the \n\nMorningags were the principal; the inferior were careful at school. \n\nTheir style is not unlike that of the Turkish potter. Here the grand aim is to obtain a common arena. . . . . To English readers of books, the present day is one of the most exciting and disappointing experiences in European literature. We carelessly allow this people to be the exclusive champions of civilization and true civilization.", - "<|bos|>NY besides that and never' 't can on' weighing ourselves. To be dead'st... . . or dreaming'st......\n\nOur'ret too comin' (2) to the ordinary, non.. \n\nFloors (1) to pleasoit person. \n\nSane (a) can amply and gratuitously as long-a.. \n\nOn' puntity, be more alive... \n\nWaiving, too secundity, let us (y) take... \n\nAre cases ofctority sufficient, sometimes, epis-ty, disELLY \n\nORKILL idea. \n\nPUR-BO", - "<|bos|>-boat built by John Eddy \n\nGreen... 623 \n\nJEFFERSON, JOHN (b. Jan. 1831), commenced business as hotel and tavern-keeper at \n\nFickenkamp, Cal., Nov. 26, 1827.. \n\n289; succeeded to business as hotel-keeper and proprietor Dec. 22, 1844.. \n\n320; commenced his business as hotel-keeper and thenceforth became a hotel and boarding-house keeper. \n\n323; successfully carried on business as hotel-keeper and thenceforth became a hotel and boarding-house keeper.. \n\n329; at end of ", - "<|bos|>. \n\nColored by Hugh Angola M'Nabbs, Commodore James E. Lightwood, Notables. This in- amidships. ventilation of the service contests in the testingroom of the national cemetery will facilitate the work among sailors who are anxious to see their fellow-patriots die.\n\nColored from a Painting, by Luella Vancouver, L.S. \n\nNoticed by Asa G-Giveno. Feather.\n\nThere are grave dangers to the hospital which must be avoided.\n\nAcres of Described by Charles. \n\nSir Francis Drake's Louisiana-Book, $ 1555-1571", - "<|bos|>. \n\n TRUSTEES.]. [The whole difference between the trust companies of his farm, Blodgett v. Nugent, 63 K. B. 521, and the trust companies of Reingeldt v. Wilbraham, 95 A. 118, 56 Am. St. Rep. 232, was merely a clear difference of intention.]\n\nIt was also clear there had been a necessary delivery of the sound overseas cart can Turnusey v. Metropolitan St., etc., R. Co., 58 L. R. A. 655, and was a clear and acquies", - "<|bos|>ING Cosmopolites. \n\nSurviving Evidence. \n\nAs to the main objects of the girl's introduction, not necessarily to a rendition of 820-44 22305 N229 Jones concludes that a report of 22-8 has been inserted in this case regarding]\n\nburying be given by me, and that it was thought that this would facilitate our future proceedings.\n\nSome attempt was made in vain to find the advertisement notificatione by defendant, but one defendant named as gruffly as the was, namely, the boy M\u00e1 vhdvr\u00e1 to groom of the horse;", - "<|bos|> gilt west, sundry small pieces of paper were found in his apartment.\n\n1762.] buoyant as lightning. This piece, preserved in a drawer in the library of the British House of Commons, was a very bad article, and not a few of the proprietor's horse fell off as it fell from him. His mother was crying, and a number of other women were taking care of their milk boxes under the bed-window. One of them stripped the stranger of his best clothes. After the fireman had been succeeded to some small articles which were every moment received with peculiar satisfaction by his buttons and clenched f" - ] -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/evals/val_bpb.json b/experiments/think-d12-1ep-44sh-r20-wd42/evals/val_bpb.json deleted file mode 100644 index ce33926da84abd0a27d2ed8d5755a60b19aefc80..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 4200)", - "step": 4200, - "bpb": { - "val": 1.0191548981297465 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/run.json b/experiments/think-d12-1ep-44sh-r20-wd42/run.json deleted file mode 100644 index 990617f6764d3b22ee9d07d0161630ea14804f40..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "stage": "base", - "base_experiment_id": "think-d12-1ep-44sh-r20-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a8d632854c2cd1bd", - "wandb_run_id": "5c4fba8a", - "created_at": 1781881918 -} diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/summary.json b/experiments/think-d12-1ep-44sh-r20-wd42/summary.json deleted file mode 100644 index ea775df32bee7a4e334fa46d42d2dd8a5b054afc..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "stage": "base", - "base_experiment_id": "think-d12-1ep-44sh-r20-wd42", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset", - "dataset_revision": "main", - "step": 4200, - "depth": 12, - "target_param_data_ratio": 20.0, - "training_tokens": 2202009600, - "final_sampled_val_bpb": 1.078128321044417, - "minimum_sampled_val_bpb": 1.078128321044417, - "full_val_bpb": 1.0191548981297465, - "core_metric": 0.07908813484502109, - "centered_results": { - "hellaswag_zeroshot": 0.04335125287373861, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.07814575731754303, - "arc_easy": 0.08810325463612874, - "arc_challenge": -0.06370876232783, - "copa": 0.13999998569488525, - "commonsense_qa": 0.14004914462566373, - "piqa": 0.10554945468902588, - "openbook_qa": -0.002666652202606201, - "lambada_openai": 0.26043081283569336, - "hellaswag": 0.04202349980672201, - "winograd": 0.1208791732788086, - "winogrande": -0.003946363925933838, - "bigbench_dyck_languages": 0.11500000208616257, - "agi_eval_lsat_ar": 0.07608695328235625, - "bigbench_cs_algorithms": 0.4015151262283325, - "bigbench_operators": 0.10476190596818924, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.025922421365976334, - "coqa": 0.0899411216378212, - "boolq": -0.20473206043243405, - "bigbench_language_identification": 0.1822882060563997 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the world. \n\nThe capital of the world is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the gold of the world. The gold of the world is the" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be the last day of the week. \n\nI am, dear Sir, your most" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the best of all. \n\nThe best of all is the best of all." - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, which is the sun of the solar system. \n\n" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the same as that of the sun, and the same as that of the moon" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the first and second, and x is the number of the second" - } - ], - "unconditioned_samples": [ - "<|bos|>IT\n\nS. PERMISSION Dedicated to the CSTRORRS OF BOSTON. \n\nADVERTISEMENT \n\nWHILE I am following up the translation of the \"Saul,\" it will not be useless to point out to the different subscribers, and to each subscriber, the different phases, phases and phases of the phenomena presented - an appearance presented in all its phases and genera, and common to them all, so as to convey to all minds then present, an impression which is correct as it is true. \n\nYet a work is what it professes to be, and this, to a very considerable extent, ought to go on for ever", - "<|bos|>900 women, or 96,000 or \n\n105,000 inhabitants, and in general the Morningags. Their occupation was not pleasant. . Some of them were married, and had children. Of these the \n\nMorningags were the principal; the inferior were careful at school. \n\nTheir style is not unlike that of the Turkish potter. Here the grand aim is to obtain a common arena. . . . . To English readers of books, the present day is one of the most exciting and disappointing experiences in European literature. We carelessly allow this people to be the exclusive champions of civilization and true civilization.", - "<|bos|>NY besides that and never' 't can on' weighing ourselves. To be dead'st... . . or dreaming'st......\n\nOur'ret too comin' (2) to the ordinary, non.. \n\nFloors (1) to pleasoit person. \n\nSane (a) can amply and gratuitously as long-a.. \n\nOn' puntity, be more alive... \n\nWaiving, too secundity, let us (y) take... \n\nAre cases ofctority sufficient, sometimes, epis-ty, disELLY \n\nORKILL idea. \n\nPUR-BO", - "<|bos|>-boat built by John Eddy \n\nGreen... 623 \n\nJEFFERSON, JOHN (b. Jan. 1831), commenced business as hotel and tavern-keeper at \n\nFickenkamp, Cal., Nov. 26, 1827.. \n\n289; succeeded to business as hotel-keeper and proprietor Dec. 22, 1844.. \n\n320; commenced his business as hotel-keeper and thenceforth became a hotel and boarding-house keeper. \n\n323; successfully carried on business as hotel-keeper and thenceforth became a hotel and boarding-house keeper.. \n\n329; at end of ", - "<|bos|>. \n\nColored by Hugh Angola M'Nabbs, Commodore James E. Lightwood, Notables. This in- amidships. ventilation of the service contests in the testingroom of the national cemetery will facilitate the work among sailors who are anxious to see their fellow-patriots die.\n\nColored from a Painting, by Luella Vancouver, L.S. \n\nNoticed by Asa G-Giveno. Feather.\n\nThere are grave dangers to the hospital which must be avoided.\n\nAcres of Described by Charles. \n\nSir Francis Drake's Louisiana-Book, $ 1555-1571", - "<|bos|>. \n\n TRUSTEES.]. [The whole difference between the trust companies of his farm, Blodgett v. Nugent, 63 K. B. 521, and the trust companies of Reingeldt v. Wilbraham, 95 A. 118, 56 Am. St. Rep. 232, was merely a clear difference of intention.]\n\nIt was also clear there had been a necessary delivery of the sound overseas cart can Turnusey v. Metropolitan St., etc., R. Co., 58 L. R. A. 655, and was a clear and acquies", - "<|bos|>ING Cosmopolites. \n\nSurviving Evidence. \n\nAs to the main objects of the girl's introduction, not necessarily to a rendition of 820-44 22305 N229 Jones concludes that a report of 22-8 has been inserted in this case regarding]\n\nburying be given by me, and that it was thought that this would facilitate our future proceedings.\n\nSome attempt was made in vain to find the advertisement notificatione by defendant, but one defendant named as gruffly as the was, namely, the boy M\u00e1 vhdvr\u00e1 to groom of the horse;", - "<|bos|> gilt west, sundry small pieces of paper were found in his apartment.\n\n1762.] buoyant as lightning. This piece, preserved in a drawer in the library of the British House of Commons, was a very bad article, and not a few of the proprietor's horse fell off as it fell from him. His mother was crying, and a number of other women were taking care of their milk boxes under the bed-window. One of them stripped the stranger of his best clothes. After the fireman had been succeeded to some small articles which were every moment received with peculiar satisfaction by his buttons and clenched f" - ], - "training_time_seconds": 11219.814347743988, - "stage_training_flops": 1.9533980655157248e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.9533980655157248e+18, - "config_fingerprint": "a8d632854c2cd1bd", - "git_commit_sha": "fd4d149a650d0fb07ae9b96864bb40e5769622d0", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/5c4fba8a", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-1ep-44sh-r20-wd42", - "dataset_fingerprint": "a6e1b3a100e0d8b3", - "tokenizer_fingerprint": "1744d7b7ee0d5d80", - "unique_train_tokens": 2268069888, - "effective_epochs": 0.970873786407767 -} diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer/experiment_tokenizer.json b/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer/experiment_tokenizer.json deleted file mode 100644 index e4983b47a9e5fee5f46c909e962afe2310ae1dbe..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "think-d12-1ep-44sh-r20-wd42", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1781881934 -} diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer/token_bytes.pt b/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer/token_bytes.pt deleted file mode 100644 index 80dbb386d071538021ab399ee6e965ae0cd1a54e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:59f928e04aa2ac37dd4064493240d1e73ecab7acb217c5a183311b0c523a3468 -size 132649 diff --git a/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer/tokenizer.pkl b/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer/tokenizer.pkl deleted file mode 100644 index a812730c3de1acc1e9e30ef6c4ccf1e9360f5d8a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-44sh-r20-wd42/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fd583e3c35851c62295a1a0f6d688923e4f30649ac963b701ec2440fe8bc3e4f -size 404221 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_000500.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_000500.json deleted file mode 100644 index 350bf0bb1e52e063da5adc52977e79f661b50c44..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 500, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.323730996039066, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.323730996039066, - "smooth_train_loss": 3.6702374931405064, - "total_training_time": 1313.3025135993958, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_001000.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_001000.json deleted file mode 100644 index 3a2f6410d80646d561db3ebc633aad6cc00f3a36..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 1000, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.253497672143533, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.253497672143533, - "smooth_train_loss": 3.404809871021335, - "total_training_time": 2657.893961429596, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_001500.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_001500.json deleted file mode 100644 index a712defb624393eb101c8daba5979a88969dc3f7..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 1500, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.2322366946944483, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.2322366946944483, - "smooth_train_loss": 3.4686871369235353, - "total_training_time": 4000.771213531494, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_002000.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_002000.json deleted file mode 100644 index 0b39b710d4315637037a45646b12a28ae3c1d613..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 2000, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.212170813932778, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.212170813932778, - "smooth_train_loss": 3.525094410637873, - "total_training_time": 5345.096604824066, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_002500.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_002500.json deleted file mode 100644 index 11bb1070a1c10a301397fe9d81cb73880cf68005..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_002500.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 2500, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.1919329110762835, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 13, - "pos": 10792769, - "epoch": 1, - "pq_idx": 13, - "rg_idx": 10792769 - }, - "loop_state": { - "min_val_bpb": 1.1919329110762835, - "smooth_train_loss": 3.4785363277537242, - "total_training_time": 6687.278959035873, - "stage_training_flops": 1162736943759360000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1162736943759360000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_003000.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_003000.json deleted file mode 100644 index 00d466b02d1cee4ab58d492c1714721d1ecd3b25..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_003000.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 3000, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.181723141516681, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 15, - "pos": 72944769, - "epoch": 1, - "pq_idx": 15, - "rg_idx": 72944769 - }, - "loop_state": { - "min_val_bpb": 1.181723141516681, - "smooth_train_loss": 3.1625074370797437, - "total_training_time": 8029.638848543167, - "stage_training_flops": 1395284332511232000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1395284332511232000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_003500.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_003500.json deleted file mode 100644 index ae6949445152310ef99e38c07adf90d8979be3d8..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_003500.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 3500, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.1594974101297872, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 18, - "pos": 35096769, - "epoch": 1, - "pq_idx": 18, - "rg_idx": 35096769 - }, - "loop_state": { - "min_val_bpb": 1.1594974101297872, - "smooth_train_loss": 3.1518777563692284, - "total_training_time": 9373.077644109726, - "stage_training_flops": 1627831721263104000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1627831721263104000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_004000.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_004000.json deleted file mode 100644 index 0aa1925d4fe0404ed0e23cd19843fe2ea034659f..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_004000.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 4000, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.1364783907083784, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 20, - "pos": 97248769, - "epoch": 1, - "pq_idx": 20, - "rg_idx": 97248769 - }, - "loop_state": { - "min_val_bpb": 1.1364783907083784, - "smooth_train_loss": 3.090057974445858, - "total_training_time": 10716.081592082977, - "stage_training_flops": 1860379110014976000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1860379110014976000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_004500.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_004500.json deleted file mode 100644 index a18bc3fc05e6e76f00d53bed2fd391962955762d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_004500.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 4500, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.1270159041981165, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 23, - "pos": 59400769, - "epoch": 1, - "pq_idx": 23, - "rg_idx": 59400769 - }, - "loop_state": { - "min_val_bpb": 1.1270159041981165, - "smooth_train_loss": 2.9730753725244097, - "total_training_time": 12057.107451677322, - "stage_training_flops": 2092926498766848000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2092926498766848000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_005000.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_005000.json deleted file mode 100644 index e4b118dbbc6d587759704a6ae9feaeb05bb343ab..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_005000.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 5000, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.1003152146678035, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 26, - "pos": 21552769, - "epoch": 1, - "pq_idx": 26, - "rg_idx": 21552769 - }, - "loop_state": { - "min_val_bpb": 1.1003152146678035, - "smooth_train_loss": 2.9552833417519717, - "total_training_time": 13398.46120429039, - "stage_training_flops": 2325473887518720000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2325473887518720000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_005500.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_005500.json deleted file mode 100644 index f23a21159bd966c66f10b6168c615c98559659b5..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_005500.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 5500, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.0836933513196012, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 28, - "pos": 83704769, - "epoch": 1, - "pq_idx": 28, - "rg_idx": 83704769 - }, - "loop_state": { - "min_val_bpb": 1.0836933513196012, - "smooth_train_loss": 2.8071836344710714, - "total_training_time": 14740.56656551361, - "stage_training_flops": 2558021276270592000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2558021276270592000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_006000.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_006000.json deleted file mode 100644 index 7f68b335bb6a082f9bd1b26bcfb5e66a345861ea..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_006000.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 6000, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.066233148422904, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 31, - "pos": 45856769, - "epoch": 1, - "pq_idx": 31, - "rg_idx": 45856769 - }, - "loop_state": { - "min_val_bpb": 1.066233148422904, - "smooth_train_loss": 2.9179841120344037, - "total_training_time": 16083.970601320267, - "stage_training_flops": 2790568665022464000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2790568665022464000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_006300.json b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_006300.json deleted file mode 100644 index e5b295c12cf7ffcff903cda2870d430aad3f06b1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/meta_006300.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "step": 6300, - "experiment_id": "think-d12-1ep-65sh-r30", - "val_bpb": 1.0599363451930062, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30", - "wandb_run_id": "6465e19b", - "wandb_group": "think-d12-stopping-point", - "wandb_tags": "think-dataset,d12,ratio30,65-shards,a100,bf16,stopping-point", - "device_type": "cuda", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 30.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "experiment_id": "think-d12-1ep-65sh-r30", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/config.json", - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-1ep-65sh-r30", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" - }, - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 33, - "pos": 3147969, - "epoch": 1, - "pq_idx": 33, - "rg_idx": 3147969 - }, - "loop_state": { - "min_val_bpb": 1.0599363451930062, - "smooth_train_loss": 3.085275168344294, - "total_training_time": 16888.896875858307, - "stage_training_flops": 2930097098273587200, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2930097098273587200 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_000500.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_000500.pt deleted file mode 100644 index 45b9aac64c47bd4a6d14749ced441afee4661403..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e7579f03e7000ac33397b4f2208abcc52e1688aad1d43912fdd4fa043337bc11 -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_001000.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_001000.pt deleted file mode 100644 index 2d71a055edd897b2d007fb0231df9b45775570b1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d57d206d7570139d91f93250c6b2f7fd871fb4a92feb5e29774d2e8e9fa0bbdc -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_001500.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_001500.pt deleted file mode 100644 index 1034e436118cd39c9531420ab7228a2f0af5d024..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2a0686bc0df28b62fd0d6469bb3a902921a54877caeeba5edb7fdf47dcdc2844 -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_002000.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_002000.pt deleted file mode 100644 index 2d815c5804be90d06d7b70d27a1eff14c169bbba..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0b29a740892304d8e9a53101c37886468d238ab248c55125985a98f2095f71f9 -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_002500.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_002500.pt deleted file mode 100644 index aab4270b26c1adb22bffc3c9876da177ffcb38bd..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:12fb590d8e7aa25e03a650c489069eda09ce374cf00eaeaa7ccf540fe3bdee40 -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_003000.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_003000.pt deleted file mode 100644 index a05309eb1274b8ae9ab6b206e8e4717899d484f8..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_003000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:188fc8b507fa7fec339848ed5149ddf07c8fcdaf958e722f8b0b172b2bc17860 -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_003500.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_003500.pt deleted file mode 100644 index 960a7749c47a760921440ae7de73e4d9282fd1e7..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_003500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:67ea1af272d836c2079acaa278d346cecf5f8dd8dd8c42550fa6bb66903726e8 -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_004000.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_004000.pt deleted file mode 100644 index e28877754d3415befdef22192af4fb3783f946e9..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_004000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3538088522149985eacbf7a5ecab7608ccaee6038e6ba82de479d8aac96e1003 -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_004500.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_004500.pt deleted file mode 100644 index 3e0f27862ab8e6e3f3816ab379a789e5a2e6902a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_004500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7b19d6094f30f368d2565a86871d3ee1f0e90591463f03b70374dfeaed304cb3 -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_005000.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_005000.pt deleted file mode 100644 index 562be79e0f8fc0336670266dc75908413380f674..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_005000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1853e0fd0989183a230b4b5c475513886a9c9d9f179f1a830c63f96052ee63bd -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_005500.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_005500.pt deleted file mode 100644 index 17cb19a2c640da74aeecf5847fce80e21efe2785..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_005500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:160d7befc0388a10fed79fd6baa8d27bfee65b467cd1d3a44554778bcaf3c8ad -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_006000.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_006000.pt deleted file mode 100644 index ac7b7cf42cd3e3507b97b5bcb7c9b0249f4a7ee4..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_006000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4ff8e4f3ce3936511384f649f32ad53acb7fd4ba2e863767b40b7d26e8a44082 -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_006300.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_006300.pt deleted file mode 100644 index 4e9930202062a2ebe7a294a549dd557ad4e19226..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/model_006300.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:72ffa92bb120cc2c27cad221a5455af61b0477254548aa37adc827c106c66f05 -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 6a6bf5d9a0d373ec30c5a46bac9a2990823d8dde..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:cc38ea6e811af742cf411b5b39039f2194ccb6c81c2be4cbaff4fb9b25ebdee8 -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 21cfe42d0a603335f84d4b648e81991a4ebeed25..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:133052ae63cbb210f05abf8ea80fa85e1d1e09b79878979c113d67c37de674f6 -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 34fd4fb645d8dad62d8bf32289a9268f5d99448c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0c923b6845bf68e534f21a06a5f9a9c389bb209851d429cf8c1829cfe5792539 -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index f0ff2ad0a840eb3951be93109c3cef09f61dd2e7..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:778f8f4d4aca74f629eca2e5a760dff9798c25c791708e221d2faa1a9d4f2e0a -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_002500_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_002500_rank0.pt deleted file mode 100644 index 68160de2f2796542242fa1c967337bd13200f691..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a9c99f66e83c7afa9d542a171f8dd658c8cabd11665208ead7a3d4ec154ea814 -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_003000_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_003000_rank0.pt deleted file mode 100644 index 59ebb4c95482d7c8e79b26f663c6343d5a1e5ea1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_003000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fe69226396ac7d8f4dccc3978108cb7be34ef50a093f45179dca0d8d40a8b109 -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_003500_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_003500_rank0.pt deleted file mode 100644 index a24d3129439b5186ec4ba909c006ffa2b89628ab..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_003500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:718f64853439ed8189ccc3290b66e3a645f659e9d971b8cff9b94f79606985d7 -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_004000_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_004000_rank0.pt deleted file mode 100644 index 6c87cc29a86ed8278da1e6f3e6f615f779c801ad..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_004000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b530816ecf7345865652f66951aefd1b4f4ff5302096a57e2afce9c50dbccf6d -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_004500_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_004500_rank0.pt deleted file mode 100644 index 62f07d1cb295dec0e10ea839d174804b6922f00e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_004500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3ac376908ab0af78ba1c69d8c088d468747da0408a8af55fe32725e88dac34ef -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_005000_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_005000_rank0.pt deleted file mode 100644 index 9d8ff04f5dd2ea372a5b737d8d43bac88d3f8783..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_005000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:afdc346269c49d35157c90ec0065626e63fdab4761d9d9ae3972496b7ce95684 -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_005500_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_005500_rank0.pt deleted file mode 100644 index 04c527a47ddeb8ce8f731fb0eff39609d3afde6a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_005500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:391f380f1d2c5adb3fa53e14eb95f47a5dfc5e43653fd6624e5c6393078a0079 -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_006000_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_006000_rank0.pt deleted file mode 100644 index b4cf70b0dd2adcd5ec12c852f45fcd56a54c9222..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_006000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e29d1a55360d980ee024a1321bf7ab5fb75b0b29d669678d2c83f102b2258694 -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_006300_rank0.pt b/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_006300_rank0.pt deleted file mode 100644 index 1b89bed251b28305fa77db256774565dd5966f85..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/base_checkpoints/optim_006300_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:848d902c15b5bb6cead5359c42c4a995dcebe24689f13867fc70331a8b84133c -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/config.json b/experiments/think-d12-1ep-65sh-r30/config.json deleted file mode 100644 index 4eaebe8677e68e025920d45a91cf6f8e307eed19..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/config.json +++ /dev/null @@ -1,74 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "target_tokens": 3402104832, - "slack": 1.03, - "require_no_wrap": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "device_type": "cuda", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "scaling_params": 110100912, - "target_param_data_ratio": 30.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-1ep-65sh-r30", - "group": "think-d12-stopping-point", - "tags": [ - "think-dataset", - "d12", - "ratio30", - "65-shards", - "a100", - "bf16", - "stopping-point" - ] - }, - "config_fingerprint": "35996219a51996ca", - "artifact_path": "experiments/think-d12-1ep-65sh-r30" -} diff --git a/experiments/think-d12-1ep-65sh-r30/evals/core.json b/experiments/think-d12-1ep-65sh-r30/evals/core.json deleted file mode 100644 index 3ee795639a139d42aeb7f52fb53c4eddf06904aa..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 6300)", - "step": 6300, - "bpb": {}, - "core_metric": 0.08955908817565156, - "core_results": { - "hellaswag_zeroshot": 0.2850029766559601, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.09463116526603699, - "arc_easy": 0.32575756311416626, - "arc_challenge": 0.20392490923404694, - "copa": 0.5899999737739563, - "commonsense_qa": 0.31859132647514343, - "piqa": 0.5631120800971985, - "openbook_qa": 0.23400001227855682, - "lambada_openai": 0.23345623910427094, - "hellaswag": 0.2857000529766083, - "winograd": 0.5494505763053894, - "winogrande": 0.5114443302154541, - "bigbench_dyck_languages": 0.11000000685453415, - "agi_eval_lsat_ar": 0.27391302585601807, - "bigbench_cs_algorithms": 0.40984848141670227, - "bigbench_operators": 0.095238097012043, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.047209080308675766, - "coqa": 0.09420017153024673, - "boolq": 0.5911315083503723, - "bigbench_language_identification": 0.25509998202323914 - }, - "centered_results": { - "hellaswag_zeroshot": 0.04667063554128011, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.09463116526603699, - "arc_easy": 0.10101008415222168, - "arc_challenge": -0.06143345435460409, - "copa": 0.1799999475479126, - "commonsense_qa": 0.14823915809392926, - "piqa": 0.12622416019439697, - "openbook_qa": -0.021333316961924236, - "lambada_openai": 0.23345623910427094, - "hellaswag": 0.0476000706354777, - "winograd": 0.09890115261077881, - "winogrande": 0.022888660430908203, - "bigbench_dyck_languages": 0.11000000685453415, - "agi_eval_lsat_ar": 0.09239128232002257, - "bigbench_cs_algorithms": 0.40984848141670227, - "bigbench_operators": 0.095238097012043, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.047209080308675766, - "coqa": 0.09420017153024673, - "boolq": -0.07596971486744127, - "bigbench_language_identification": 0.18052803302886594 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/results.json b/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/results.json deleted file mode 100644 index ed623e7c775eb06c128ed3e30ce20af801e2a154..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/results.json +++ /dev/null @@ -1,224 +0,0 @@ -{ - "experiment_id": "think-d12-1ep-65sh-r30", - "note": "Checkpoint results share one ratio-30 learning-rate schedule; they scout candidate regions and are not independent ratio runs.", - "records": [ - { - "step": 2500, - "realized_ratio": 11.90471519436642, - "training_tokens": 1310720000, - "stage_training_flops": 1.16273694375936e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.16273694375936e+18, - "core_metric": 0.047198013376400304, - "full_val_bpb": 1.1436998415921598, - "centered_results": { - "hellaswag_zeroshot": 0.031932552655537925, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.05545986816287041, - "arc_easy": 0.0628507137298584, - "arc_challenge": -0.054607510566711426, - "copa": 0.019999980926513672, - "commonsense_qa": 0.09807536751031874, - "piqa": 0.06093573570251465, - "openbook_qa": -0.013333320617675781, - "lambada_openai": 0.23423248529434204, - "hellaswag": 0.030737559000651043, - "winograd": 0.040293097496032715, - "winogrande": 0.035516977310180664, - "bigbench_dyck_languages": 0.10700000822544098, - "agi_eval_lsat_ar": 0.09239128232002257, - "bigbench_cs_algorithms": 0.3742424249649048, - "bigbench_operators": 0.05238095298409462, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.004824976436793804, - "coqa": 0.04509582743048668, - "boolq": -0.4155802758116471, - "bigbench_language_identification": 0.17590759112627724 - }, - "output_json": "step_002500.json" - }, - { - "step": 3500, - "realized_ratio": 16.666601272112988, - "training_tokens": 1835008000, - "stage_training_flops": 1.627831721263104e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.627831721263104e+18, - "core_metric": 0.06822591677690754, - "full_val_bpb": 1.1049592886593402, - "centered_results": { - "hellaswag_zeroshot": 0.03511913617451986, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.05482013523578644, - "arc_easy": 0.08754209677378337, - "arc_challenge": -0.06712174415588379, - "copa": 0.059999942779541016, - "commonsense_qa": 0.08579032868146895, - "piqa": 0.06202387809753418, - "openbook_qa": -0.03999998172124227, - "lambada_openai": 0.18144769966602325, - "hellaswag": 0.032463630040486656, - "winograd": 0.09890115261077881, - "winogrande": -0.041831135749816895, - "bigbench_dyck_languages": 0.11300000548362732, - "agi_eval_lsat_ar": 0.05434781685471533, - "bigbench_cs_algorithms": 0.40984848141670227, - "bigbench_operators": 0.11428572237491608, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.017975401133298874, - "coqa": 0.08204935491085052, - "boolq": -0.02446487075404116, - "bigbench_language_identification": 0.18382838614309582 - }, - "output_json": "step_003500.json" - }, - { - "step": 4000, - "realized_ratio": 19.04754431098627, - "training_tokens": 2097152000, - "stage_training_flops": 1.860379110014976e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.860379110014976e+18, - "core_metric": 0.06920160601195241, - "full_val_bpb": 1.0842222791064016, - "centered_results": { - "hellaswag_zeroshot": 0.03392418225606283, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.0632350742816925, - "arc_easy": 0.0740740696589152, - "arc_challenge": -0.06484641631444295, - "copa": 0.019999980926513672, - "commonsense_qa": 0.10319410264492034, - "piqa": 0.08922743797302246, - "openbook_qa": -0.010666648546854654, - "lambada_openai": 0.20395885407924652, - "hellaswag": 0.03578305244445801, - "winograd": 0.1208791732788086, - "winogrande": 0.03709542751312256, - "bigbench_dyck_languages": 0.09200000762939453, - "agi_eval_lsat_ar": 0.05978258699178694, - "bigbench_cs_algorithms": 0.3787878751754761, - "bigbench_operators": 0.0714285746216774, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.022421948611736298, - "coqa": 0.0712764635682106, - "boolq": -0.0622887234938772, - "bigbench_language_identification": 0.1831683089630832 - }, - "output_json": "step_004000.json" - }, - { - "step": 4500, - "realized_ratio": 21.428487349859555, - "training_tokens": 2359296000, - "stage_training_flops": 2.092926498766848e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2.092926498766848e+18, - "core_metric": 0.07931023722789925, - "full_val_bpb": 1.0642527211292854, - "centered_results": { - "hellaswag_zeroshot": 0.03697800636291504, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.07760444283485413, - "arc_easy": 0.07800225416819255, - "arc_challenge": -0.0443686048189799, - "copa": 0.12000000476837158, - "commonsense_qa": 0.11445535719394682, - "piqa": 0.10228502750396729, - "openbook_qa": -0.013333320617675781, - "lambada_openai": 0.19871918857097626, - "hellaswag": 0.03445525964101156, - "winograd": 0.16483521461486816, - "winogrande": 0.014996051788330078, - "bigbench_dyck_languages": 0.08900000154972076, - "agi_eval_lsat_ar": 0.07608695328235625, - "bigbench_cs_algorithms": 0.4280302822589874, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.021002838388085365, - "coqa": 0.05837404355406761, - "boolq": -0.07033644851885343, - "bigbench_language_identification": 0.18184818738889116 - }, - "output_json": "step_004500.json" - }, - { - "step": 5500, - "realized_ratio": 26.190373427606122, - "training_tokens": 2883584000, - "stage_training_flops": 2.558021276270592e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2.558021276270592e+18, - "core_metric": 0.08353097318258858, - "full_val_bpb": 1.01586582084303, - "centered_results": { - "hellaswag_zeroshot": 0.0462723175684611, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.05792037770152092, - "arc_easy": 0.0892255703608195, - "arc_challenge": -0.0443686048189799, - "copa": 0.2200000286102295, - "commonsense_qa": 0.11547910422086714, - "piqa": 0.13710546493530273, - "openbook_qa": -0.010666648546854654, - "lambada_openai": 0.20531728863716125, - "hellaswag": 0.043484012285868325, - "winograd": 0.11355316638946533, - "winogrande": -0.005524873733520508, - "bigbench_dyck_languages": 0.10000000149011612, - "agi_eval_lsat_ar": 0.08695649355649947, - "bigbench_cs_algorithms": 0.3946969509124756, - "bigbench_operators": 0.08571428805589676, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.042478714138269424, - "coqa": 0.0854315385222435, - "boolq": -0.1105746030807495, - "bigbench_language_identification": 0.18470845626394608 - }, - "output_json": "step_005500.json" - }, - { - "step": 6300, - "realized_ratio": 29.999882289803377, - "training_tokens": 3303014400, - "stage_training_flops": 2.930097098273587e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2.930097098273587e+18, - "core_metric": 0.08955908817565156, - "full_val_bpb": 0.9944563924677391, - "centered_results": { - "hellaswag_zeroshot": 0.04667063554128011, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.09463116526603699, - "arc_easy": 0.10101008415222168, - "arc_challenge": -0.06143345435460409, - "copa": 0.1799999475479126, - "commonsense_qa": 0.14823915809392926, - "piqa": 0.12622416019439697, - "openbook_qa": -0.021333316961924236, - "lambada_openai": 0.23345623910427094, - "hellaswag": 0.0476000706354777, - "winograd": 0.09890115261077881, - "winogrande": 0.022888660430908203, - "bigbench_dyck_languages": 0.11000000685453415, - "agi_eval_lsat_ar": 0.09239128232002257, - "bigbench_cs_algorithms": 0.40984848141670227, - "bigbench_operators": 0.095238097012043, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.047209080308675766, - "coqa": 0.09420017153024673, - "boolq": -0.07596971486744127, - "bigbench_language_identification": 0.18052803302886594 - }, - "output_json": "step_006300.json" - } - ], - "wandb_logged_steps": [ - 2500, - 3500, - 4000, - 4500, - 5500, - 6300 - ] -} diff --git a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_002500.json b/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_002500.json deleted file mode 100644 index 7ebbd9946c101ccdec2d48c09bfbc91aa0ccebe3..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_002500.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "model": "base_model (step 2500)", - "step": 2500, - "bpb": { - "val": 1.1436998415921598 - }, - "core_metric": 0.047198013376400304, - "core_results": { - "hellaswag_zeroshot": 0.27394941449165344, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.05545986816287041, - "arc_easy": 0.2971380352973938, - "arc_challenge": 0.20904436707496643, - "copa": 0.5099999904632568, - "commonsense_qa": 0.278460294008255, - "piqa": 0.5304678678512573, - "openbook_qa": 0.24000000953674316, - "lambada_openai": 0.23423248529434204, - "hellaswag": 0.2730531692504883, - "winograd": 0.5201465487480164, - "winogrande": 0.5177584886550903, - "bigbench_dyck_languages": 0.10700000822544098, - "agi_eval_lsat_ar": 0.27391302585601807, - "bigbench_cs_algorithms": 0.3742424249649048, - "bigbench_operators": 0.05238095298409462, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.004824976436793804, - "coqa": 0.04509582743048668, - "boolq": 0.4620794951915741, - "bigbench_language_identification": 0.250900000333786 - }, - "centered_results": { - "hellaswag_zeroshot": 0.031932552655537925, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.05545986816287041, - "arc_easy": 0.0628507137298584, - "arc_challenge": -0.054607510566711426, - "copa": 0.019999980926513672, - "commonsense_qa": 0.09807536751031874, - "piqa": 0.06093573570251465, - "openbook_qa": -0.013333320617675781, - "lambada_openai": 0.23423248529434204, - "hellaswag": 0.030737559000651043, - "winograd": 0.040293097496032715, - "winogrande": 0.035516977310180664, - "bigbench_dyck_languages": 0.10700000822544098, - "agi_eval_lsat_ar": 0.09239128232002257, - "bigbench_cs_algorithms": 0.3742424249649048, - "bigbench_operators": 0.05238095298409462, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.004824976436793804, - "coqa": 0.04509582743048668, - "boolq": -0.4155802758116471, - "bigbench_language_identification": 0.17590759112627724 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_003500.json b/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_003500.json deleted file mode 100644 index a20e8124002e8642932f54cf17d088e5d29cba06..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_003500.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "model": "base_model (step 3500)", - "step": 3500, - "bpb": { - "val": 1.1049592886593402 - }, - "core_metric": 0.06822591677690754, - "core_results": { - "hellaswag_zeroshot": 0.2763393521308899, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.05482013523578644, - "arc_easy": 0.3156565725803375, - "arc_challenge": 0.19965869188308716, - "copa": 0.5299999713897705, - "commonsense_qa": 0.26863226294517517, - "piqa": 0.5310119390487671, - "openbook_qa": 0.2200000137090683, - "lambada_openai": 0.18144769966602325, - "hellaswag": 0.274347722530365, - "winograd": 0.5494505763053894, - "winogrande": 0.47908443212509155, - "bigbench_dyck_languages": 0.11300000548362732, - "agi_eval_lsat_ar": 0.24347825348377228, - "bigbench_cs_algorithms": 0.40984848141670227, - "bigbench_operators": 0.11428572237491608, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.017975401133298874, - "coqa": 0.08204935491085052, - "boolq": 0.6107033491134644, - "bigbench_language_identification": 0.2581000030040741 - }, - "centered_results": { - "hellaswag_zeroshot": 0.03511913617451986, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.05482013523578644, - "arc_easy": 0.08754209677378337, - "arc_challenge": -0.06712174415588379, - "copa": 0.059999942779541016, - "commonsense_qa": 0.08579032868146895, - "piqa": 0.06202387809753418, - "openbook_qa": -0.03999998172124227, - "lambada_openai": 0.18144769966602325, - "hellaswag": 0.032463630040486656, - "winograd": 0.09890115261077881, - "winogrande": -0.041831135749816895, - "bigbench_dyck_languages": 0.11300000548362732, - "agi_eval_lsat_ar": 0.05434781685471533, - "bigbench_cs_algorithms": 0.40984848141670227, - "bigbench_operators": 0.11428572237491608, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.017975401133298874, - "coqa": 0.08204935491085052, - "boolq": -0.02446487075404116, - "bigbench_language_identification": 0.18382838614309582 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_004000.json b/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_004000.json deleted file mode 100644 index d47c6c34d45aadb6d697075867eb62c5b8ef8fb6..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_004000.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "model": "base_model (step 4000)", - "step": 4000, - "bpb": { - "val": 1.0842222791064016 - }, - "core_metric": 0.06920160601195241, - "core_results": { - "hellaswag_zeroshot": 0.2754431366920471, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.0632350742816925, - "arc_easy": 0.3055555522441864, - "arc_challenge": 0.20136518776416779, - "copa": 0.5099999904632568, - "commonsense_qa": 0.2825552821159363, - "piqa": 0.5446137189865112, - "openbook_qa": 0.242000013589859, - "lambada_openai": 0.20395885407924652, - "hellaswag": 0.2768372893333435, - "winograd": 0.5604395866394043, - "winogrande": 0.5185477137565613, - "bigbench_dyck_languages": 0.09200000762939453, - "agi_eval_lsat_ar": 0.24782606959342957, - "bigbench_cs_algorithms": 0.3787878751754761, - "bigbench_operators": 0.0714285746216774, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.022421948611736298, - "coqa": 0.0712764635682106, - "boolq": 0.5963302850723267, - "bigbench_language_identification": 0.2574999928474426 - }, - "centered_results": { - "hellaswag_zeroshot": 0.03392418225606283, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.0632350742816925, - "arc_easy": 0.0740740696589152, - "arc_challenge": -0.06484641631444295, - "copa": 0.019999980926513672, - "commonsense_qa": 0.10319410264492034, - "piqa": 0.08922743797302246, - "openbook_qa": -0.010666648546854654, - "lambada_openai": 0.20395885407924652, - "hellaswag": 0.03578305244445801, - "winograd": 0.1208791732788086, - "winogrande": 0.03709542751312256, - "bigbench_dyck_languages": 0.09200000762939453, - "agi_eval_lsat_ar": 0.05978258699178694, - "bigbench_cs_algorithms": 0.3787878751754761, - "bigbench_operators": 0.0714285746216774, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.022421948611736298, - "coqa": 0.0712764635682106, - "boolq": -0.0622887234938772, - "bigbench_language_identification": 0.1831683089630832 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_004500.json b/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_004500.json deleted file mode 100644 index 58f98943514f7954ea03f5e0e836dc44d2d44c8e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_004500.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "model": "base_model (step 4500)", - "step": 4500, - "bpb": { - "val": 1.0642527211292854 - }, - "core_metric": 0.07931023722789925, - "core_results": { - "hellaswag_zeroshot": 0.2777335047721863, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.07760444283485413, - "arc_easy": 0.3085016906261444, - "arc_challenge": 0.21672354638576508, - "copa": 0.5600000023841858, - "commonsense_qa": 0.29156428575515747, - "piqa": 0.5511425137519836, - "openbook_qa": 0.24000000953674316, - "lambada_openai": 0.19871918857097626, - "hellaswag": 0.27584144473075867, - "winograd": 0.5824176073074341, - "winogrande": 0.507498025894165, - "bigbench_dyck_languages": 0.08900000154972076, - "agi_eval_lsat_ar": 0.260869562625885, - "bigbench_cs_algorithms": 0.4280302822589874, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.021002838388085365, - "coqa": 0.05837404355406761, - "boolq": 0.5932721495628357, - "bigbench_language_identification": 0.2563000023365021 - }, - "centered_results": { - "hellaswag_zeroshot": 0.03697800636291504, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.07760444283485413, - "arc_easy": 0.07800225416819255, - "arc_challenge": -0.0443686048189799, - "copa": 0.12000000476837158, - "commonsense_qa": 0.11445535719394682, - "piqa": 0.10228502750396729, - "openbook_qa": -0.013333320617675781, - "lambada_openai": 0.19871918857097626, - "hellaswag": 0.03445525964101156, - "winograd": 0.16483521461486816, - "winogrande": 0.014996051788330078, - "bigbench_dyck_languages": 0.08900000154972076, - "agi_eval_lsat_ar": 0.07608695328235625, - "bigbench_cs_algorithms": 0.4280302822589874, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.021002838388085365, - "coqa": 0.05837404355406761, - "boolq": -0.07033644851885343, - "bigbench_language_identification": 0.18184818738889116 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_005500.json b/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_005500.json deleted file mode 100644 index cea1e83ec7520e54334006dd5c299afec7e8619b..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_005500.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "model": "base_model (step 5500)", - "step": 5500, - "bpb": { - "val": 1.01586582084303 - }, - "core_metric": 0.08353097318258858, - "core_results": { - "hellaswag_zeroshot": 0.2847042381763458, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.05792037770152092, - "arc_easy": 0.3169191777706146, - "arc_challenge": 0.21672354638576508, - "copa": 0.6100000143051147, - "commonsense_qa": 0.2923832833766937, - "piqa": 0.5685527324676514, - "openbook_qa": 0.242000013589859, - "lambada_openai": 0.20531728863716125, - "hellaswag": 0.28261300921440125, - "winograd": 0.5567765831947327, - "winogrande": 0.49723756313323975, - "bigbench_dyck_languages": 0.10000000149011612, - "agi_eval_lsat_ar": 0.2695651948451996, - "bigbench_cs_algorithms": 0.3946969509124756, - "bigbench_operators": 0.08571428805589676, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.042478714138269424, - "coqa": 0.0854315385222435, - "boolq": 0.5779816508293152, - "bigbench_language_identification": 0.258899986743927 - }, - "centered_results": { - "hellaswag_zeroshot": 0.0462723175684611, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.05792037770152092, - "arc_easy": 0.0892255703608195, - "arc_challenge": -0.0443686048189799, - "copa": 0.2200000286102295, - "commonsense_qa": 0.11547910422086714, - "piqa": 0.13710546493530273, - "openbook_qa": -0.010666648546854654, - "lambada_openai": 0.20531728863716125, - "hellaswag": 0.043484012285868325, - "winograd": 0.11355316638946533, - "winogrande": -0.005524873733520508, - "bigbench_dyck_languages": 0.10000000149011612, - "agi_eval_lsat_ar": 0.08695649355649947, - "bigbench_cs_algorithms": 0.3946969509124756, - "bigbench_operators": 0.08571428805589676, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.042478714138269424, - "coqa": 0.0854315385222435, - "boolq": -0.1105746030807495, - "bigbench_language_identification": 0.18470845626394608 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_006300.json b/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_006300.json deleted file mode 100644 index a84ed823369661c99327c4327b268811d3577b38..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/evals/ratio_scout/step_006300.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "model": "base_model (step 6300)", - "step": 6300, - "bpb": { - "val": 0.9944563924677391 - }, - "core_metric": 0.08955908817565156, - "core_results": { - "hellaswag_zeroshot": 0.2850029766559601, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.09463116526603699, - "arc_easy": 0.32575756311416626, - "arc_challenge": 0.20392490923404694, - "copa": 0.5899999737739563, - "commonsense_qa": 0.31859132647514343, - "piqa": 0.5631120800971985, - "openbook_qa": 0.23400001227855682, - "lambada_openai": 0.23345623910427094, - "hellaswag": 0.2857000529766083, - "winograd": 0.5494505763053894, - "winogrande": 0.5114443302154541, - "bigbench_dyck_languages": 0.11000000685453415, - "agi_eval_lsat_ar": 0.27391302585601807, - "bigbench_cs_algorithms": 0.40984848141670227, - "bigbench_operators": 0.095238097012043, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.047209080308675766, - "coqa": 0.09420017153024673, - "boolq": 0.5911315083503723, - "bigbench_language_identification": 0.25509998202323914 - }, - "centered_results": { - "hellaswag_zeroshot": 0.04667063554128011, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.09463116526603699, - "arc_easy": 0.10101008415222168, - "arc_challenge": -0.06143345435460409, - "copa": 0.1799999475479126, - "commonsense_qa": 0.14823915809392926, - "piqa": 0.12622416019439697, - "openbook_qa": -0.021333316961924236, - "lambada_openai": 0.23345623910427094, - "hellaswag": 0.0476000706354777, - "winograd": 0.09890115261077881, - "winogrande": 0.022888660430908203, - "bigbench_dyck_languages": 0.11000000685453415, - "agi_eval_lsat_ar": 0.09239128232002257, - "bigbench_cs_algorithms": 0.40984848141670227, - "bigbench_operators": 0.095238097012043, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.047209080308675766, - "coqa": 0.09420017153024673, - "boolq": -0.07596971486744127, - "bigbench_language_identification": 0.18052803302886594 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} diff --git a/experiments/think-d12-1ep-65sh-r30/evals/samples.json b/experiments/think-d12-1ep-65sh-r30/evals/samples.json deleted file mode 100644 index 188dc944e0e4f8b4afb1f72eeba9ea0b6dca771d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 6300)", - "step": 6300, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the world. \n\nThe capital of the world is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as that of silver, and the same as that of copper. The" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nI am, dear Sir, Your most obedient servant,\n\nJ." - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is hot, and the hot is cold. \n\nThe hot is hot, and the" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, and the stars. \n\n2." - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the white, and the white is the color of the black. \n\nThe white" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the first pair of eyes, and x is the number of the" - } - ], - "unconditioned_samples": [ - "<|bos|>LEES, BARNS & W sale,12R.\n\nLONDON: I. SONS, 68-72 MARKET STREET; LEEDS,-PL ZAN Bauth & Co., 30 WALNUT STREET; ME., 1-69; \n\n12-70 Market STREET.\n\nEduc 7622.11 Vennifths \n\nCOLLEGE HARVARD HAR HAR LIBRARY FROM RR.Harov to 1913 \n\nUnivied by fiv end.\n\nman, Latham; The Cabbies' Tales, Raiders, Experiences, Myters and Other \n\nMarks, Andes); Hymns, Odd Fellows, Odd \n\nFellows", - "<|bos|>ily engaged in common domestic life again, until, weary of living in tents, restricted within narrow bounds, at the ATLstock farm, where they were accustomed to meet and converse in common, it was too much for them to expect to endure ichkosisa realities; they became aware of the fact, and having come together for worship, it was a melodramatic performance.\n\nThe Inviciosorum closer system. \n\nAnd now, humous clay workings, whose steep walls have stood the free\u03c4\u03af\u03b6\u03c9\u03bd into the world for so many hundreds of generations, begin to rise with the dirt instigated by the witch", - "<|bos|>HARVARD Edinburgh, and approved of Bland in presenting certain designs which he designed for queen's theatres and building churches, or accomplished very singular effects in external art as man served us, and concerning which we the English rose musically, and saidtherefore like most of her modernit say, probably from the bust of Miss V., and endeavour to praise the happy conduct of the French' pacificists, in being able to perpetuat, and within her theatre, towards this ment, as this very compliment was used to her with talent, though in lists for every day relative liberally; for, her night-attentresses. \n\n", - "<|bos|>Oliver ORmonds. \n\n666.] A river may cut a dozen bars of iron of different forms and shapes; it may do so with sundry branches, and not with one uniform hard iron line or one uniform thickness; it may construct a timber stump of a jutting sort of oaken beam, or of eights, cylinders, or 25\"4 square inches extending from two pine branches, upon one road-stone, in one frame and without frame, and then 3\" through, etc.\"-Walter Bevis. \n\n592.] No intermediate plank, or wrought-iron frame at all :- -", - "<|bos|> \u05d9\u05d4 \u05d5\u05d4 \n\nINSTITVTIO THEOLOGICA ANDOVER FVNDATA MDCCCVII \n\n\u0391\u039a\u03a1\u039f\u0393\u03a9\u039d \n\nPs.CXIX JOH.XVII. \n\n179. oyocate\u2758 in- \u05d4\u05e0\u05d5\u05d3\u05d9 \u03cc\u03c3\u03bf\u03c2 \n\nChairs in Ps.CXIX JOH.XVII. \n\n169. \u05db\u05e8\u05db\u05e8\u05da $ 5.00\n\nMy address to Esther.\n\nY. H. C.\n\nJAMES STEPHEN Jeannette \n\nRichard Bladen, Onkelos' own act-died in 1863.\n\nJAMES STEPHEN Jeannette, York which he left to be the issue of the marriage of General Graham with Julia Morgan (my brother-in-law, the countess of \n\nB", - "<|bos|>alogue be transacted. The first instant vessel approximating to loading at the moment of port signals should be shaped.\" Her heel must be anchored relatively low in the snow.\" \n\nSo soon as \"papers to sell\" may claim permission to enter ports, Livingston refused to organise the commercialering business.\" \"Syrup of violets and pinks,\u201d adds Fairhaven, \"are infrequent people, and are not much used by those that live in cities.\"7 \"Darting in Position,\" \"ida every thing except its periodical patterns.\" \"The pace and The smooth back guile", - "<|bos|> cemetery mills. . I sunk the Hudson's\n\nRiver steamer at Hamilton, New York, and telegraphed to the chairman of the board of directors 210 articles; disposed of by me at 100 each; disposed of at 50; only one lot taken off by accident occurred, and some irritated by language supposed to have been used! A skirmish occurred near Schenectady, N. Y., between the parties at 50; Grant, in person, made one of the battles; \n\nHal The Better Hill since called Home Hill; was the battle ground of General to-day.\n\nI was in the", - "<|bos|>ventions extensive Government insularffitho-anatolaruced.\n\nWhen we were still intestinely in the midst of #1 \n\nMahyrr' and not surprising the Oles they 460 this family | and the most eran wide Simpson.HISTORY AND \n\nour sety 18sn had magst hon and aour way if fagles Su8 f the | led their formed 66 must maintain she cler that the hand of iod st bomb e ideparting the une art ells buso ear and guts instde tep Our puff eth like the Jachers" - ] -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/evals/val_bpb.json b/experiments/think-d12-1ep-65sh-r30/evals/val_bpb.json deleted file mode 100644 index 6f31ea31f32a5068a07fd4f78170dc17af903ab1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 6300)", - "step": 6300, - "bpb": { - "val": 0.9944563924677391 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/run.json b/experiments/think-d12-1ep-65sh-r30/run.json deleted file mode 100644 index 02d6e7a2f1647cb3b8b242b78f106b76fd3a5368..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-1ep-65sh-r30", - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "35996219a51996ca", - "wandb_run_id": "6465e19b", - "created_at": 1781533685 -} diff --git a/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/checkpoints/meta_000015.json b/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/checkpoints/meta_000015.json deleted file mode 100644 index eed1fd5664678a38c6a6360502c766fca9b99aca..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/checkpoints/meta_000015.json +++ /dev/null @@ -1,103 +0,0 @@ -{ - "step": 15, - "training_complete": true, - "val_bpb": 0.9086840222563478, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-1ep-65sh-r30-pre1930-authentic", - "wandb_run_id": "22d50a96", - "wandb_group": "think-d12", - "wandb_tags": "sft,pre1930,ratio20", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/base_checkpoints", - "base_step": 6300, - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/checkpoints", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/tokenizer", - "resume_from_step": null, - "experiment_id": "think-d12-1ep-65sh-r30-pre1930-authentic", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/config.json", - "parent_cumulative_flops": 2.930097098273587e+18, - "tokenizer_fingerprint": "db3bec0946e70097", - "git_commit_sha": "4f8ea13fe1030b32a2fa9cdf4b4bad215e22790e", - "load_optimizer": 1, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.0, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": -1, - "eval_tokens": 20971520, - "chatcore_every": -1, - "chatcore_max_cat": -1, - "chatcore_max_sample": 24, - "save_every": -1, - "recipe": "pre1930", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-authentic", - "data": { - "recipe": "pre1930", - "pre1930_epochs": 5 - }, - "training": { - "num_iterations": -1, - "device_batch_size": 8, - "eval_every": -1, - "chatcore_every": -1, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "pre1930", - "ratio20" - ] - }, - "config_fingerprint": "d3378357cef17359", - "artifact_path": "experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "d3378357cef17359" - }, - "loop_state": { - "step": 15, - "total_training_time": 64.28930902481079, - "min_val_bpb": 0.9086840222563478, - "smooth_train_loss": 1.518641630821301, - "mfu": 29.814679341911354, - "tok_per_sec": 40667, - "stage_training_flops": 6976421662556160.0, - "inherited_parent_flops": 2.930097098273587e+18, - "cumulative_pipeline_training_flops": 2.9370735199361434e+18 - } -} \ No newline at end of file diff --git a/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/checkpoints/model_000015.pt b/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/checkpoints/model_000015.pt deleted file mode 100644 index 86db94b0f10763d783fc55ff553ee880971bc438..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/checkpoints/model_000015.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a67f28fea4d8c23b765f975a5adff79ea59dbbca91be51ffde858f3131298d6b -size 792761690 diff --git a/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/checkpoints/optim_000015_rank0.pt b/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/checkpoints/optim_000015_rank0.pt deleted file mode 100644 index 09c137247ed9221174cf4a980061b59f36fcd375..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/checkpoints/optim_000015_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2c5868bc12712d790e251a9e4e0e13bac2a629e8e1a32fd81cc9daf3a5d3a6bb -size 1246165357 diff --git a/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/config.json b/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/config.json deleted file mode 100644 index 5b8d21cabaa018bda6bb6442d064afa773a87740..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/config.json +++ /dev/null @@ -1,32 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-authentic", - "data": { - "recipe": "pre1930", - "pre1930_epochs": 5 - }, - "training": { - "num_iterations": -1, - "device_batch_size": 8, - "eval_every": -1, - "chatcore_every": -1, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "pre1930", - "ratio20" - ] - }, - "config_fingerprint": "d3378357cef17359", - "artifact_path": "experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic" -} diff --git a/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/run.json b/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/run.json deleted file mode 100644 index 5a112a9c675ea2f0dbb4d36b617b2077d821801d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/sft/think-d12-1ep-65sh-r30-pre1930-authentic/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-1ep-65sh-r30-pre1930-authentic", - "stage": "sft", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": "think-d12-1ep-65sh-r30", - "parent_checkpoint_step": null, - "config_fingerprint": "d3378357cef17359", - "wandb_run_id": "22d50a96", - "created_at": 1782333681 -} diff --git a/experiments/think-d12-1ep-65sh-r30/summary.json b/experiments/think-d12-1ep-65sh-r30/summary.json deleted file mode 100644 index 7f653387e8878689c1d499f42f6732ae31f54cbb..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "think-d12-1ep-65sh-r30", - "stage": "base", - "base_experiment_id": "think-d12-1ep-65sh-r30", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset", - "dataset_revision": "main", - "step": 6300, - "depth": 12, - "target_param_data_ratio": 30.0, - "training_tokens": 3303014400, - "final_sampled_val_bpb": 1.0599363451930062, - "minimum_sampled_val_bpb": 1.0599363451930062, - "full_val_bpb": 0.9944563924677391, - "core_metric": 0.08955908817565156, - "centered_results": { - "hellaswag_zeroshot": 0.04667063554128011, - "jeopardy": 0.0, - "bigbench_qa_wikidata": 0.09463116526603699, - "arc_easy": 0.10101008415222168, - "arc_challenge": -0.06143345435460409, - "copa": 0.1799999475479126, - "commonsense_qa": 0.14823915809392926, - "piqa": 0.12622416019439697, - "openbook_qa": -0.021333316961924236, - "lambada_openai": 0.23345623910427094, - "hellaswag": 0.0476000706354777, - "winograd": 0.09890115261077881, - "winogrande": 0.022888660430908203, - "bigbench_dyck_languages": 0.11000000685453415, - "agi_eval_lsat_ar": 0.09239128232002257, - "bigbench_cs_algorithms": 0.40984848141670227, - "bigbench_operators": 0.095238097012043, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.047209080308675766, - "coqa": 0.09420017153024673, - "boolq": -0.07596971486744127, - "bigbench_language_identification": 0.18052803302886594 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the world. \n\nThe capital of the world is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as that of silver, and the same as that of copper. The" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nI am, dear Sir, Your most obedient servant,\n\nJ." - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is hot, and the hot is cold. \n\nThe hot is hot, and the" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, and the stars. \n\n2." - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the white, and the white is the color of the black. \n\nThe white" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the first pair of eyes, and x is the number of the" - } - ], - "unconditioned_samples": [ - "<|bos|>LEES, BARNS & W sale,12R.\n\nLONDON: I. SONS, 68-72 MARKET STREET; LEEDS,-PL ZAN Bauth & Co., 30 WALNUT STREET; ME., 1-69; \n\n12-70 Market STREET.\n\nEduc 7622.11 Vennifths \n\nCOLLEGE HARVARD HAR HAR LIBRARY FROM RR.Harov to 1913 \n\nUnivied by fiv end.\n\nman, Latham; The Cabbies' Tales, Raiders, Experiences, Myters and Other \n\nMarks, Andes); Hymns, Odd Fellows, Odd \n\nFellows", - "<|bos|>ily engaged in common domestic life again, until, weary of living in tents, restricted within narrow bounds, at the ATLstock farm, where they were accustomed to meet and converse in common, it was too much for them to expect to endure ichkosisa realities; they became aware of the fact, and having come together for worship, it was a melodramatic performance.\n\nThe Inviciosorum closer system. \n\nAnd now, humous clay workings, whose steep walls have stood the free\u03c4\u03af\u03b6\u03c9\u03bd into the world for so many hundreds of generations, begin to rise with the dirt instigated by the witch", - "<|bos|>HARVARD Edinburgh, and approved of Bland in presenting certain designs which he designed for queen's theatres and building churches, or accomplished very singular effects in external art as man served us, and concerning which we the English rose musically, and saidtherefore like most of her modernit say, probably from the bust of Miss V., and endeavour to praise the happy conduct of the French' pacificists, in being able to perpetuat, and within her theatre, towards this ment, as this very compliment was used to her with talent, though in lists for every day relative liberally; for, her night-attentresses. \n\n", - "<|bos|>Oliver ORmonds. \n\n666.] A river may cut a dozen bars of iron of different forms and shapes; it may do so with sundry branches, and not with one uniform hard iron line or one uniform thickness; it may construct a timber stump of a jutting sort of oaken beam, or of eights, cylinders, or 25\"4 square inches extending from two pine branches, upon one road-stone, in one frame and without frame, and then 3\" through, etc.\"-Walter Bevis. \n\n592.] No intermediate plank, or wrought-iron frame at all :- -", - "<|bos|> \u05d9\u05d4 \u05d5\u05d4 \n\nINSTITVTIO THEOLOGICA ANDOVER FVNDATA MDCCCVII \n\n\u0391\u039a\u03a1\u039f\u0393\u03a9\u039d \n\nPs.CXIX JOH.XVII. \n\n179. oyocate\u2758 in- \u05d4\u05e0\u05d5\u05d3\u05d9 \u03cc\u03c3\u03bf\u03c2 \n\nChairs in Ps.CXIX JOH.XVII. \n\n169. \u05db\u05e8\u05db\u05e8\u05da $ 5.00\n\nMy address to Esther.\n\nY. H. C.\n\nJAMES STEPHEN Jeannette \n\nRichard Bladen, Onkelos' own act-died in 1863.\n\nJAMES STEPHEN Jeannette, York which he left to be the issue of the marriage of General Graham with Julia Morgan (my brother-in-law, the countess of \n\nB", - "<|bos|>alogue be transacted. The first instant vessel approximating to loading at the moment of port signals should be shaped.\" Her heel must be anchored relatively low in the snow.\" \n\nSo soon as \"papers to sell\" may claim permission to enter ports, Livingston refused to organise the commercialering business.\" \"Syrup of violets and pinks,\u201d adds Fairhaven, \"are infrequent people, and are not much used by those that live in cities.\"7 \"Darting in Position,\" \"ida every thing except its periodical patterns.\" \"The pace and The smooth back guile", - "<|bos|> cemetery mills. . I sunk the Hudson's\n\nRiver steamer at Hamilton, New York, and telegraphed to the chairman of the board of directors 210 articles; disposed of by me at 100 each; disposed of at 50; only one lot taken off by accident occurred, and some irritated by language supposed to have been used! A skirmish occurred near Schenectady, N. Y., between the parties at 50; Grant, in person, made one of the battles; \n\nHal The Better Hill since called Home Hill; was the battle ground of General to-day.\n\nI was in the", - "<|bos|>ventions extensive Government insularffitho-anatolaruced.\n\nWhen we were still intestinely in the midst of #1 \n\nMahyrr' and not surprising the Oles they 460 this family | and the most eran wide Simpson.HISTORY AND \n\nour sety 18sn had magst hon and aour way if fagles Su8 f the | led their formed 66 must maintain she cler that the hand of iod st bomb e ideparting the une art ells buso ear and guts instde tep Our puff eth like the Jachers" - ], - "training_time_seconds": 16888.896875858307, - "stage_training_flops": 2.930097098273587e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2.930097098273587e+18, - "config_fingerprint": "35996219a51996ca", - "git_commit_sha": "09b5f3b7b61a45f34d50a36e570b85f0259208bc", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/6465e19b", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-1ep-65sh-r30", - "dataset_fingerprint": "6749d397ed8a3e8a", - "tokenizer_fingerprint": "db3bec0946e70097", - "unique_train_tokens": 3402104832, - "effective_epochs": 0.970873786407767 -} diff --git a/experiments/think-d12-1ep-65sh-r30/tokenizer/experiment_tokenizer.json b/experiments/think-d12-1ep-65sh-r30/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 21337c125185ad7d836eb9f28cd0e89d03e6447b..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "think-d12-1ep-65sh-r30", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 65, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1781533715 -} diff --git a/experiments/think-d12-1ep-65sh-r30/tokenizer/token_bytes.pt b/experiments/think-d12-1ep-65sh-r30/tokenizer/token_bytes.pt deleted file mode 100644 index 3b5650014a721f951ecce07734b0c3bd85fef590..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f4d99f73dc52d95e87da791073a41423fe2230fa2555cd32868244cb49717311 -size 132649 diff --git a/experiments/think-d12-1ep-65sh-r30/tokenizer/tokenizer.pkl b/experiments/think-d12-1ep-65sh-r30/tokenizer/tokenizer.pkl deleted file mode 100644 index e1fbfc56da70168a3f04052fedef3a8d3e48baf9..0000000000000000000000000000000000000000 --- a/experiments/think-d12-1ep-65sh-r30/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7fb5b395796a23910af92c571853fd4265eb22fb852c73378d329c6852780865 -size 404114 diff --git a/experiments/think-d12-r11-run1/base_checkpoints/meta_000500.json b/experiments/think-d12-r11-run1/base_checkpoints/meta_000500.json deleted file mode 100644 index dedee2b0686096115c0d465b68433457fd2e8ee3..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 500, - "experiment_id": "think-d12-r11-run1", - "val_bpb": 1.3230782875794327, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run1", - "wandb_run_id": "5ffd9e1f", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/base_checkpoints", - "experiment_id": "think-d12-r11-run1", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run1/config.json", - "tokenizer_fingerprint": "5a7fc542f0b39fb9", - "git_commit_sha": "4d4b7c5e2a8f77393c2c541587f4cd311b7122e0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run1", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "56aa9474a0ebe6a9", - "artifact_path": "experiments/think-d12-r11-run1" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "56aa9474a0ebe6a9" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3230782875794327, - "smooth_train_loss": 3.5873454787650205, - "total_training_time": 1301.10826253891, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run1/base_checkpoints/meta_001000.json b/experiments/think-d12-r11-run1/base_checkpoints/meta_001000.json deleted file mode 100644 index 6c233b1ec823bc1d9e0812e8fcf077bedd53a64d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 1000, - "experiment_id": "think-d12-r11-run1", - "val_bpb": 1.2327391515334245, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run1", - "wandb_run_id": "5ffd9e1f", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/base_checkpoints", - "experiment_id": "think-d12-r11-run1", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run1/config.json", - "tokenizer_fingerprint": "5a7fc542f0b39fb9", - "git_commit_sha": "4d4b7c5e2a8f77393c2c541587f4cd311b7122e0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run1", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "56aa9474a0ebe6a9", - "artifact_path": "experiments/think-d12-r11-run1" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "56aa9474a0ebe6a9" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2327391515334245, - "smooth_train_loss": 3.4437952294455405, - "total_training_time": 2629.6326158046722, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run1/base_checkpoints/meta_001500.json b/experiments/think-d12-r11-run1/base_checkpoints/meta_001500.json deleted file mode 100644 index 8b77f551f274f4303c75d07a7465269e407a1032..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 1500, - "experiment_id": "think-d12-r11-run1", - "val_bpb": 1.1752708057117873, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run1", - "wandb_run_id": "5ffd9e1f", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/base_checkpoints", - "experiment_id": "think-d12-r11-run1", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run1/config.json", - "tokenizer_fingerprint": "5a7fc542f0b39fb9", - "git_commit_sha": "4d4b7c5e2a8f77393c2c541587f4cd311b7122e0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run1", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "56aa9474a0ebe6a9", - "artifact_path": "experiments/think-d12-r11-run1" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "56aa9474a0ebe6a9" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.1752708057117873, - "smooth_train_loss": 3.2445274349683744, - "total_training_time": 3959.332376718521, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run1/base_checkpoints/meta_002000.json b/experiments/think-d12-r11-run1/base_checkpoints/meta_002000.json deleted file mode 100644 index c05bbc4aab592d317243f0f82e0c320f42fe3bf6..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 2000, - "experiment_id": "think-d12-r11-run1", - "val_bpb": 1.1234441835089963, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run1", - "wandb_run_id": "5ffd9e1f", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/base_checkpoints", - "experiment_id": "think-d12-r11-run1", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run1/config.json", - "tokenizer_fingerprint": "5a7fc542f0b39fb9", - "git_commit_sha": "4d4b7c5e2a8f77393c2c541587f4cd311b7122e0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run1", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "56aa9474a0ebe6a9", - "artifact_path": "experiments/think-d12-r11-run1" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "56aa9474a0ebe6a9" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.1234441835089963, - "smooth_train_loss": 3.2435862129271023, - "total_training_time": 5287.725782871246, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run1/base_checkpoints/meta_002362.json b/experiments/think-d12-r11-run1/base_checkpoints/meta_002362.json deleted file mode 100644 index 204a2e60f7fc67675a6c02166ffdded054299299..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 2362, - "experiment_id": "think-d12-r11-run1", - "val_bpb": 1.1026519310163985, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run1", - "wandb_run_id": "5ffd9e1f", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run1/base_checkpoints", - "experiment_id": "think-d12-r11-run1", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run1/config.json", - "tokenizer_fingerprint": "5a7fc542f0b39fb9", - "git_commit_sha": "4d4b7c5e2a8f77393c2c541587f4cd311b7122e0", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run1", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run1", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "56aa9474a0ebe6a9", - "artifact_path": "experiments/think-d12-r11-run1" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "56aa9474a0ebe6a9" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": 1.1026519310163985, - "smooth_train_loss": 3.073952729045518, - "total_training_time": 6249.3527302742, - "stage_training_flops": 1098553864463843328, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1098553864463843328 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run1/base_checkpoints/model_000500.pt b/experiments/think-d12-r11-run1/base_checkpoints/model_000500.pt deleted file mode 100644 index 68a45ed0848cf9788d5d9f1e989e44c3a134972c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f163e0f3998bd8d58a3484d103ce027036032d9a205b5a3ef38864f557fb1fc5 -size 792761690 diff --git a/experiments/think-d12-r11-run1/base_checkpoints/model_001000.pt b/experiments/think-d12-r11-run1/base_checkpoints/model_001000.pt deleted file mode 100644 index e2f872cc4369fd8999f15d7e35f4d0eac38c2a28..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a962593d363a9329bffb69a38ffc18152eb473c2bf2c93257342ab5df8af24c8 -size 792761690 diff --git a/experiments/think-d12-r11-run1/base_checkpoints/model_001500.pt b/experiments/think-d12-r11-run1/base_checkpoints/model_001500.pt deleted file mode 100644 index d4c95927dcab99fb99fb1a191f8ff1a72794f0db..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:caa2774af1ed018d45043681fa8062532c12182b9e9c998b0b8db34c0dc62363 -size 792761690 diff --git a/experiments/think-d12-r11-run1/base_checkpoints/model_002000.pt b/experiments/think-d12-r11-run1/base_checkpoints/model_002000.pt deleted file mode 100644 index 6029e2c269da256f74ca42faeedb7fe6c9e8fece..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bdebf84a71e02ee2a66c76258a1371c6e3cbcba232b2fd2477b4782b13e8dea4 -size 792761690 diff --git a/experiments/think-d12-r11-run1/base_checkpoints/model_002362.pt b/experiments/think-d12-r11-run1/base_checkpoints/model_002362.pt deleted file mode 100644 index c4d4d98f0ed88996891da9ae87b843eed02738cb..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8822f407b4d43983fc9d4585d3b8ed85563940ba00b23a2490f3006d3776a282 -size 792761690 diff --git a/experiments/think-d12-r11-run1/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r11-run1/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 0dceefc2b0cffdb5716c96b5879baaabf287dfe9..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:cf215b8cb2fc13fe79a03eebb40e2e6f322227892b2fd6212aa9891eb40a4611 -size 1246165357 diff --git a/experiments/think-d12-r11-run1/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r11-run1/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 48c1ec3403cbbe18d9a88cf98fbec2eb521455eb..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4ff8f324df287c17f0ca6985b499fac453f2d28f66219755e6ab39ebd1ee5e5c -size 1246165357 diff --git a/experiments/think-d12-r11-run1/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r11-run1/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index c6cfd7975d8a783f8730be57df056a822b4e9a37..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3d744e7957459bb2e1ccdd5f3fc5932a4fc62d04072d531adfd8932353e22bb7 -size 1246165357 diff --git a/experiments/think-d12-r11-run1/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r11-run1/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 268e84fa10143716258b18fc5e6802ce74cbf758..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d2950ece9fc33659f8ee2a229d702ecce584fceb294345a14a0e5f26721fab1e -size 1246165357 diff --git a/experiments/think-d12-r11-run1/base_checkpoints/optim_002362_rank0.pt b/experiments/think-d12-r11-run1/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 5b147787e2a81471964a57ba7166a3fdea511b16..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:660b1d1a545db50aa1a158248270e509c4605d9824f5b3b7332a6b09c01f8141 -size 1246165357 diff --git a/experiments/think-d12-r11-run1/config.json b/experiments/think-d12-r11-run1/config.json deleted file mode 100644 index 52e314508da60da3652457f24c73494f07cd2602..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/config.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run1", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "56aa9474a0ebe6a9", - "artifact_path": "experiments/think-d12-r11-run1" -} diff --git a/experiments/think-d12-r11-run1/evals/core.json b/experiments/think-d12-r11-run1/evals/core.json deleted file mode 100644 index 639e54e5cc10296713f06cb6d2182a6f7cf2a1ad..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": 0.0791723549916879, - "core_results": { - "hellaswag_zeroshot": 0.2768372893333435, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.08277151733636856, - "arc_easy": 0.31186866760253906, - "arc_challenge": 0.20392490923404694, - "copa": 0.550000011920929, - "commonsense_qa": 0.31449630856513977, - "piqa": 0.5484221577644348, - "openbook_qa": 0.26200002431869507, - "lambada_openai": 0.24956335127353668, - "hellaswag": 0.28062137961387634, - "winograd": 0.5494505763053894, - "winogrande": 0.5019731521606445, - "bigbench_dyck_languages": 0.10900000482797623, - "agi_eval_lsat_ar": 0.27391302585601807, - "bigbench_cs_algorithms": 0.4219696819782257, - "bigbench_operators": 0.06666667014360428, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.02071901597082615, - "coqa": 0.07027433067560196, - "boolq": 0.5798165202140808, - "bigbench_language_identification": 0.25209999084472656 - }, - "centered_results": { - "hellaswag_zeroshot": 0.03578305244445801, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.08277151733636856, - "arc_easy": 0.08249155680338542, - "arc_challenge": -0.06143345435460409, - "copa": 0.10000002384185791, - "commonsense_qa": 0.14312038570642469, - "piqa": 0.09684431552886963, - "openbook_qa": 0.016000032424926758, - "lambada_openai": 0.24956335127353668, - "hellaswag": 0.04082850615183512, - "winograd": 0.09890115261077881, - "winogrande": 0.0039463043212890625, - "bigbench_dyck_languages": 0.10900000482797623, - "agi_eval_lsat_ar": 0.09239128232002257, - "bigbench_cs_algorithms": 0.4219696819782257, - "bigbench_operators": 0.06666667014360428, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.02071901597082615, - "coqa": 0.07027433067560196, - "boolq": -0.10574599943662943, - "bigbench_language_identification": 0.17722771270046928 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run1/evals/samples.json b/experiments/think-d12-r11-run1/evals/samples.json deleted file mode 100644 index ecaf279459ce24e7a254328edb339c9a5fe30d74..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the United States. \n\nThe capital of the United States is the" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the power of the sun. The symbol of the power of the" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nI am, my dear Sir, your most obedient servant,\n\nJOHN" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the opposite of cold. \n\nThe opposite of cold is the opposite of heat." - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, and the stars. \n\n2." - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the color of the sky, and the color of the earth. \n\nThe color" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the same as 2, and the same as 2, and the same" - } - ], - "unconditioned_samples": [ - "<|bos|>HIERONYMUS. \n\nANDOVER-HARVARD THEOLOGICAL LIBRARY M. D. \n\n1906\n\n\u0f5d \u0f0b 70 .75 13\n\nce\n\n3\n\nDead Nature's Fire \n\nTens of coat between \u25a1 different degrees of m are; peannbs meteureux differing Raphaelite sagacity. * c missavit besides sejur de stala.\"-Mr. Herschel reported that the sun has no capability in colors, and He says: \u2191 \u221e: Single sometimes sing near together a body of the same size and appearance as the oaken staff of the sun, as if vocal in color, not", - "<|bos|>370 \n\nPaceouddy river. \n\nOne pious canoe, one-To make a dozen different animals. \n\nTwo three elephants. \n\nOne a trunk in blossom, one in spring; but two different empty eggs. 73\n\n1748] \n\nPARATIAN EPIPHANES \n\nOne large flock of dogs, one sheep, A camel in spring; one little bird, one bear. \n\nTwo jackal at three angles, one at three; one bold capmaker at one see regular exercise. \n\nTwo preserved sprigs of honey and an apple, with four dessert and melons.", - "<|bos|>Frrep rep\u00faitentot's excuse on the Si les.\"-Mure in Kerridge's Essays, Note B.-B. Blanquet. By Amyot.--A true account of the publicity, editing, and journeys of the parties \n\nWith Mayanned Notes. - FROM THE LIFE OF LEOPOLD DE GONTIER. Selected \n\n&c. &c. Printed for J. O. and R. Reston, London, 1824.\n\nF. SEIN, & CO. MCMILES, PRINTERS, NATIONAL ELECTROTYPED AND PUBLISHED. \n\n364 pp., lieving\n\nHALE COLLEGE LIBRARY \n\nDECEMBER ", - "<|bos|>HENRY MARTINEAU FOREMAN THE PERSECUTED SCHOOL \n\nMarch 23 \n\nSilk-festing the Anatomy of the Mohegans \n\nJuly 15 \n\nLetter from Mr. ALICE FROST, the Rev. Dr. SAMUEL FARNSWORTH, the Rev. Gouverneur \n\nin confidence with St. Mary Evangelist, and Dr. Seth \n\nMarch 4 Error in books, timental Library, for (Georgics, I.) 1751 page 4....... 3.......... 93 1201.)\n\nMODERN knows lies paper modesty, taking a lead to preserve us from read-\n\nooms inaccuracy in silence", - "<|bos|> HOUSE OF THE FIRST RIOTS. \n\nAgnosticism.] One who hopes to get a reputation for know- \n\n1 My Last in a Garden. I reserve my criticism of the boy in far less compatible terms than in ancient philosophy one who has hitherto trusted himself to generals. Humanitarians, by means of unphilosophical instruction, are better able to work miracles out of scepticism, scepticism, than sentimental men whose attitude towards religion could no longer be tolerated. \n\nThe lasting consequence is, that ignorance, which survives long, goes far behind the doctrine of miracle. The emotional crisis may actually be reached by absolute faith and resignation on the part", - "<|bos|>GOD AND HIS RACE. By FRANK KING NOTES FIRST NEW MEXICO. By G. H. CARY COOLIDGE \n\n(8) With an Introduction by J. F. EGAN. New York, 1900:\n\nTHE FISHERMAN. He appears in a Primer which has been compiled from \n\nProf. Hovenden's book of Professor Smith's First Principles of Geology for Research in Geology. With an Introduction by R. R. CLAY, LL.D., President of the Royal Institution. New York: \n\n1902:\n\nAN INTRODUCTION TO THE STUDY OF GEOLOGY. \n\nBy Various Authors, with a Commentary by Arthur O The Queen and her Character", - "<|bos|>Army Neglected: and Arkansas Army \n\nCoach. Vincy Vincy, Baltimore. 1.-2\n\n135 \n\nArmy\n\nMarch 1.-It was dreadfully unpopular with her people. Premising was a much dreaded term amongst them, it was asserted that she would place her confidence in little better hands than Gordon.\n\nWhat she regarded as dangerous was the thinly clad slave called \"Chasseur,\" another nominee. The trouble of conciliation was only occasionally perceptible, and, as we shall see, the Southern States refused to listen to her proposition. So that soon the situation became such that they had", - "<|bos|>the eight (1) of W\u3001 lerie w conferring constructe them-Laura de la \n\nChant, von ipol\u00f2 never' and never assures give all he has ever seen this prid\u00e8, e.'d uentechem. up half ator sau 119) 3 la raffment, non fu consolt8 f the three triumphs raf 8' i alone uaugrat student nonsense (af 9:1)!\" \u2022 cable fu \n\n12 W e barbicot, alki manton apparently improvement (nushe, waers f" - ] -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run1/evals/val_bpb.json b/experiments/think-d12-r11-run1/evals/val_bpb.json deleted file mode 100644 index 991cfe4811c5f1c9ef28f5204e86aff644f58563..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.051975107450783 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run1/run.json b/experiments/think-d12-r11-run1/run.json deleted file mode 100644 index fa4db302da7e9afc548dfb559acbcddf053171ec..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-r11-run1", - "stage": "base", - "base_experiment_id": "think-d12-r11-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "56aa9474a0ebe6a9", - "wandb_run_id": "5ffd9e1f", - "created_at": 1781891761 -} diff --git a/experiments/think-d12-r11-run1/summary.json b/experiments/think-d12-r11-run1/summary.json deleted file mode 100644 index f6d732a33c6672414394c897c0d5f1a0d6b717de..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "think-d12-r11-run1", - "stage": "base", - "base_experiment_id": "think-d12-r11-run1", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.1026519310163985, - "minimum_sampled_val_bpb": 1.1026519310163985, - "full_val_bpb": 1.051975107450783, - "core_metric": 0.0791723549916879, - "centered_results": { - "hellaswag_zeroshot": 0.03578305244445801, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.08277151733636856, - "arc_easy": 0.08249155680338542, - "arc_challenge": -0.06143345435460409, - "copa": 0.10000002384185791, - "commonsense_qa": 0.14312038570642469, - "piqa": 0.09684431552886963, - "openbook_qa": 0.016000032424926758, - "lambada_openai": 0.24956335127353668, - "hellaswag": 0.04082850615183512, - "winograd": 0.09890115261077881, - "winogrande": 0.0039463043212890625, - "bigbench_dyck_languages": 0.10900000482797623, - "agi_eval_lsat_ar": 0.09239128232002257, - "bigbench_cs_algorithms": 0.4219696819782257, - "bigbench_operators": 0.06666667014360428, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.02071901597082615, - "coqa": 0.07027433067560196, - "boolq": -0.10574599943662943, - "bigbench_language_identification": 0.17722771270046928 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the United States. \n\nThe capital of the United States is the" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the power of the sun. The symbol of the power of the" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nI am, my dear Sir, your most obedient servant,\n\nJOHN" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the opposite of cold. \n\nThe opposite of cold is the opposite of heat." - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, and the stars. \n\n2." - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the color of the sky, and the color of the earth. \n\nThe color" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the same as 2, and the same as 2, and the same" - } - ], - "unconditioned_samples": [ - "<|bos|>HIERONYMUS. \n\nANDOVER-HARVARD THEOLOGICAL LIBRARY M. D. \n\n1906\n\n\u0f5d \u0f0b 70 .75 13\n\nce\n\n3\n\nDead Nature's Fire \n\nTens of coat between \u25a1 different degrees of m are; peannbs meteureux differing Raphaelite sagacity. * c missavit besides sejur de stala.\"-Mr. Herschel reported that the sun has no capability in colors, and He says: \u2191 \u221e: Single sometimes sing near together a body of the same size and appearance as the oaken staff of the sun, as if vocal in color, not", - "<|bos|>370 \n\nPaceouddy river. \n\nOne pious canoe, one-To make a dozen different animals. \n\nTwo three elephants. \n\nOne a trunk in blossom, one in spring; but two different empty eggs. 73\n\n1748] \n\nPARATIAN EPIPHANES \n\nOne large flock of dogs, one sheep, A camel in spring; one little bird, one bear. \n\nTwo jackal at three angles, one at three; one bold capmaker at one see regular exercise. \n\nTwo preserved sprigs of honey and an apple, with four dessert and melons.", - "<|bos|>Frrep rep\u00faitentot's excuse on the Si les.\"-Mure in Kerridge's Essays, Note B.-B. Blanquet. By Amyot.--A true account of the publicity, editing, and journeys of the parties \n\nWith Mayanned Notes. - FROM THE LIFE OF LEOPOLD DE GONTIER. Selected \n\n&c. &c. Printed for J. O. and R. Reston, London, 1824.\n\nF. SEIN, & CO. MCMILES, PRINTERS, NATIONAL ELECTROTYPED AND PUBLISHED. \n\n364 pp., lieving\n\nHALE COLLEGE LIBRARY \n\nDECEMBER ", - "<|bos|>HENRY MARTINEAU FOREMAN THE PERSECUTED SCHOOL \n\nMarch 23 \n\nSilk-festing the Anatomy of the Mohegans \n\nJuly 15 \n\nLetter from Mr. ALICE FROST, the Rev. Dr. SAMUEL FARNSWORTH, the Rev. Gouverneur \n\nin confidence with St. Mary Evangelist, and Dr. Seth \n\nMarch 4 Error in books, timental Library, for (Georgics, I.) 1751 page 4....... 3.......... 93 1201.)\n\nMODERN knows lies paper modesty, taking a lead to preserve us from read-\n\nooms inaccuracy in silence", - "<|bos|> HOUSE OF THE FIRST RIOTS. \n\nAgnosticism.] One who hopes to get a reputation for know- \n\n1 My Last in a Garden. I reserve my criticism of the boy in far less compatible terms than in ancient philosophy one who has hitherto trusted himself to generals. Humanitarians, by means of unphilosophical instruction, are better able to work miracles out of scepticism, scepticism, than sentimental men whose attitude towards religion could no longer be tolerated. \n\nThe lasting consequence is, that ignorance, which survives long, goes far behind the doctrine of miracle. The emotional crisis may actually be reached by absolute faith and resignation on the part", - "<|bos|>GOD AND HIS RACE. By FRANK KING NOTES FIRST NEW MEXICO. By G. H. CARY COOLIDGE \n\n(8) With an Introduction by J. F. EGAN. New York, 1900:\n\nTHE FISHERMAN. He appears in a Primer which has been compiled from \n\nProf. Hovenden's book of Professor Smith's First Principles of Geology for Research in Geology. With an Introduction by R. R. CLAY, LL.D., President of the Royal Institution. New York: \n\n1902:\n\nAN INTRODUCTION TO THE STUDY OF GEOLOGY. \n\nBy Various Authors, with a Commentary by Arthur O The Queen and her Character", - "<|bos|>Army Neglected: and Arkansas Army \n\nCoach. Vincy Vincy, Baltimore. 1.-2\n\n135 \n\nArmy\n\nMarch 1.-It was dreadfully unpopular with her people. Premising was a much dreaded term amongst them, it was asserted that she would place her confidence in little better hands than Gordon.\n\nWhat she regarded as dangerous was the thinly clad slave called \"Chasseur,\" another nominee. The trouble of conciliation was only occasionally perceptible, and, as we shall see, the Southern States refused to listen to her proposition. So that soon the situation became such that they had", - "<|bos|>the eight (1) of W\u3001 lerie w conferring constructe them-Laura de la \n\nChant, von ipol\u00f2 never' and never assures give all he has ever seen this prid\u00e8, e.'d uentechem. up half ator sau 119) 3 la raffment, non fu consolt8 f the three triumphs raf 8' i alone uaugrat student nonsense (af 9:1)!\" \u2022 cable fu \n\n12 W e barbicot, alki manton apparently improvement (nushe, waers f" - ], - "training_time_seconds": 6249.3527302742, - "stage_training_flops": 1.0985538644638433e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.0985538644638433e+18, - "config_fingerprint": "56aa9474a0ebe6a9", - "git_commit_sha": "4d4b7c5e2a8f77393c2c541587f4cd311b7122e0", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/5ffd9e1f", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r11-run1", - "dataset_fingerprint": "63a5e6be81591d82", - "tokenizer_fingerprint": "5a7fc542f0b39fb9", - "unique_train_tokens": 1275519304, - "effective_epochs": 0.970873786164196 -} diff --git a/experiments/think-d12-r11-run1/tokenizer/experiment_tokenizer.json b/experiments/think-d12-r11-run1/tokenizer/experiment_tokenizer.json deleted file mode 100644 index d2ae30a5932f947ce26b87eca8fc3bbb83ddf0cf..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "think-d12-r11-run1", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1781890485 -} diff --git a/experiments/think-d12-r11-run1/tokenizer/token_bytes.pt b/experiments/think-d12-r11-run1/tokenizer/token_bytes.pt deleted file mode 100644 index 01d1ec4aab9e8a7d205c3b3ffbeb8da0e9a62db1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1 -size 132649 diff --git a/experiments/think-d12-r11-run1/tokenizer/tokenizer.pkl b/experiments/think-d12-r11-run1/tokenizer/tokenizer.pkl deleted file mode 100644 index a17bd392980021628053b95d6425fc556aad527a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run1/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1 -size 404071 diff --git a/experiments/think-d12-r11-run2/base_checkpoints/meta_000500.json b/experiments/think-d12-r11-run2/base_checkpoints/meta_000500.json deleted file mode 100644 index 123fa7415b3bf8a9adc7826f66f4e1747d962624..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 500, - "experiment_id": "think-d12-r11-run2", - "val_bpb": 1.3256252804681699, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run2", - "wandb_run_id": "84e31c8e", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/base_checkpoints", - "experiment_id": "think-d12-r11-run2", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run2/config.json", - "tokenizer_fingerprint": "96dd8e502a849d59", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 43, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run2", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run2", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run2", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "7f54bf07c5352d97", - "artifact_path": "experiments/think-d12-r11-run2" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "7f54bf07c5352d97" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3256252804681699, - "smooth_train_loss": 3.588497145795441, - "total_training_time": 1296.8187935352325, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run2/base_checkpoints/meta_001000.json b/experiments/think-d12-r11-run2/base_checkpoints/meta_001000.json deleted file mode 100644 index 6c17fb2048fec39c11242179c5324a14ad293552..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 1000, - "experiment_id": "think-d12-r11-run2", - "val_bpb": 1.2351307837396184, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run2", - "wandb_run_id": "84e31c8e", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/base_checkpoints", - "experiment_id": "think-d12-r11-run2", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run2/config.json", - "tokenizer_fingerprint": "96dd8e502a849d59", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 43, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run2", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run2", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run2", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "7f54bf07c5352d97", - "artifact_path": "experiments/think-d12-r11-run2" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "7f54bf07c5352d97" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2351307837396184, - "smooth_train_loss": 3.4494649735291096, - "total_training_time": 2619.1334071159363, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run2/base_checkpoints/meta_001500.json b/experiments/think-d12-r11-run2/base_checkpoints/meta_001500.json deleted file mode 100644 index d3d846e146e47cb84191d7938ffef78dfd8e2a87..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 1500, - "experiment_id": "think-d12-r11-run2", - "val_bpb": 1.1760284474435965, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run2", - "wandb_run_id": "84e31c8e", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/base_checkpoints", - "experiment_id": "think-d12-r11-run2", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run2/config.json", - "tokenizer_fingerprint": "96dd8e502a849d59", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 43, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run2", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run2", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run2", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "7f54bf07c5352d97", - "artifact_path": "experiments/think-d12-r11-run2" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "7f54bf07c5352d97" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.1760284474435965, - "smooth_train_loss": 3.247025462772539, - "total_training_time": 3941.2222929000854, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run2/base_checkpoints/meta_002000.json b/experiments/think-d12-r11-run2/base_checkpoints/meta_002000.json deleted file mode 100644 index d160329610ec65c61ca10114fa1e213f9086213f..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 2000, - "experiment_id": "think-d12-r11-run2", - "val_bpb": 1.1227598491207411, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run2", - "wandb_run_id": "84e31c8e", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/base_checkpoints", - "experiment_id": "think-d12-r11-run2", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run2/config.json", - "tokenizer_fingerprint": "96dd8e502a849d59", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 43, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run2", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run2", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run2", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "7f54bf07c5352d97", - "artifact_path": "experiments/think-d12-r11-run2" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "7f54bf07c5352d97" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.1227598491207411, - "smooth_train_loss": 3.2449127117831824, - "total_training_time": 5264.500639915466, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run2/base_checkpoints/meta_002362.json b/experiments/think-d12-r11-run2/base_checkpoints/meta_002362.json deleted file mode 100644 index c5f7d41e8667fdf1a46a2bd5c43ff0f38cba7601..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 2362, - "experiment_id": "think-d12-r11-run2", - "val_bpb": 1.101818437651454, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run2", - "wandb_run_id": "84e31c8e", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run2/base_checkpoints", - "experiment_id": "think-d12-r11-run2", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run2/config.json", - "tokenizer_fingerprint": "96dd8e502a849d59", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 43, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run2", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run2", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run2", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "7f54bf07c5352d97", - "artifact_path": "experiments/think-d12-r11-run2" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "7f54bf07c5352d97" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": 1.101818437651454, - "smooth_train_loss": 3.0713499916611764, - "total_training_time": 6222.97850894928, - "stage_training_flops": 1098553864463843328, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1098553864463843328 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run2/base_checkpoints/model_000500.pt b/experiments/think-d12-r11-run2/base_checkpoints/model_000500.pt deleted file mode 100644 index 75fc52d3ea97dd5967ef987cc73fc50b736926b4..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6aa80a8d2fbc587741198baa1c23797e9a03abf47e82daa8402749cbba01ed03 -size 792761690 diff --git a/experiments/think-d12-r11-run2/base_checkpoints/model_001000.pt b/experiments/think-d12-r11-run2/base_checkpoints/model_001000.pt deleted file mode 100644 index 0a304b79f0a6c2d2dd7919b1cf729d48ceb2f121..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4068e780a404f26a487f94bb746300d3640c41ce752520be014c3518ec09e8d1 -size 792761690 diff --git a/experiments/think-d12-r11-run2/base_checkpoints/model_001500.pt b/experiments/think-d12-r11-run2/base_checkpoints/model_001500.pt deleted file mode 100644 index 2b7d99a758bacccd3d57e3dd4bbfeae90d43da86..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:afbbdc53ee7105f7e13c54cb309e88b6f1ba8a07722ba671a4e28dc0052c6f92 -size 792761690 diff --git a/experiments/think-d12-r11-run2/base_checkpoints/model_002000.pt b/experiments/think-d12-r11-run2/base_checkpoints/model_002000.pt deleted file mode 100644 index 1080733e27e63e648d3edf94c313321989b5a699..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:ffc92e32294ed34f72e7bfb6fc4968eaec6df567a60723d8ad4cccf4b7228d4b -size 792761690 diff --git a/experiments/think-d12-r11-run2/base_checkpoints/model_002362.pt b/experiments/think-d12-r11-run2/base_checkpoints/model_002362.pt deleted file mode 100644 index 13a26e86b3c1f375010c13664f2d6798a61004a2..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:48333070fea388243792d978e6a4727bc984eb0115dd735a9de5893f8aaa8e6f -size 792761690 diff --git a/experiments/think-d12-r11-run2/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r11-run2/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index b82bd99cea63ac980474477f59f4dedb44466802..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:56797f52a92fbb0a9945f2fa451bdf957f6118037452b23f009acce5067c80c4 -size 1246165357 diff --git a/experiments/think-d12-r11-run2/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r11-run2/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 3e0b9b1b09dc953fe8b21bc723cf0907440a1045..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6601227b18f462e666bfd581050d81675e9df2d1cb0d2a0c8342c788ae16fced -size 1246165357 diff --git a/experiments/think-d12-r11-run2/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r11-run2/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 210fdfcaa0869930eeaf6068136348cfeebafe09..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b5e3172a659b7e58ffd0be60d94c7f6ea2ca39924158b4fddfc0389f7a39ef5e -size 1246165357 diff --git a/experiments/think-d12-r11-run2/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r11-run2/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 4bc87ad14282e99f19824cfd23a102392d962b51..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b419389821092e526cb0a61153dcab4ebf42277a9d36a9406c7606bb790eb619 -size 1246165357 diff --git a/experiments/think-d12-r11-run2/base_checkpoints/optim_002362_rank0.pt b/experiments/think-d12-r11-run2/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 6db65fbe3e595029fbe27a4a39f765184c37b30e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c08b338b4adcd6f31408e290e73eb41e37994c7a1dfca0aaa584f6fd0064c7de -size 1246165357 diff --git a/experiments/think-d12-r11-run2/config.json b/experiments/think-d12-r11-run2/config.json deleted file mode 100644 index a9540c3aaf44e16cf6d8327a67f0d6b7d7c3bacd..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/config.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run2", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 43, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run2", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "7f54bf07c5352d97", - "artifact_path": "experiments/think-d12-r11-run2" -} diff --git a/experiments/think-d12-r11-run2/evals/core.json b/experiments/think-d12-r11-run2/evals/core.json deleted file mode 100644 index 57b3d7328323f1a22a14afe7390062b96de7932e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": 0.06687854185240662, - "core_results": { - "hellaswag_zeroshot": 0.274347722530365, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.07002607733011246, - "arc_easy": 0.3055555522441864, - "arc_challenge": 0.21672354638576508, - "copa": 0.5299999713897705, - "commonsense_qa": 0.315315306186676, - "piqa": 0.5424374341964722, - "openbook_qa": 0.23800000548362732, - "lambada_openai": 0.19406171143054962, - "hellaswag": 0.2761402130126953, - "winograd": 0.5494505763053894, - "winogrande": 0.48382002115249634, - "bigbench_dyck_languages": 0.1120000034570694, - "agi_eval_lsat_ar": 0.25217390060424805, - "bigbench_cs_algorithms": 0.3810606002807617, - "bigbench_operators": 0.11428572237491608, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.014569535851478577, - "coqa": 0.08042089641094208, - "boolq": 0.5547400712966919, - "bigbench_language_identification": 0.24949999153614044 - }, - "centered_results": { - "hellaswag_zeroshot": 0.032463630040486656, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.07002607733011246, - "arc_easy": 0.0740740696589152, - "arc_challenge": -0.0443686048189799, - "copa": 0.059999942779541016, - "commonsense_qa": 0.144144132733345, - "piqa": 0.08487486839294434, - "openbook_qa": -0.015999992688496906, - "lambada_openai": 0.19406171143054962, - "hellaswag": 0.034853617350260414, - "winograd": 0.09890115261077881, - "winogrande": -0.032359957695007324, - "bigbench_dyck_languages": 0.1120000034570694, - "agi_eval_lsat_ar": 0.06521737575531004, - "bigbench_cs_algorithms": 0.3810606002807617, - "bigbench_operators": 0.11428572237491608, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.014569535851478577, - "coqa": 0.08042089641094208, - "boolq": -0.17173665448238973, - "bigbench_language_identification": 0.17436742743249772 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run2/evals/samples.json b/experiments/think-d12-r11-run2/evals/samples.json deleted file mode 100644 index 306690ba6c68d1f8408e0285e62ab890674610b0..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the French nation, and the capital of the French nation, the" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as that of the sun, and the same as that of the moon" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and the day of the week will be the day of the week." - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as cold. \n\nThe latter is the same as the former, and" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, and the planets. \n\n2." - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a red, and I am a little afraid of it. I have been in" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the days of the week, and the number of days of the" - } - ], - "unconditioned_samples": [ - "<|bos|>adjustments. A concern called industrial, they say, is the glory of the Modern Cabinet, and when not mentioned to history, he would be scarcely deserving to be thought a relation.\n\nMr. President, though different from Hume and Lange, we have been met by Dr. Blair and Dr. Hawke, who asked me whether they knew that his father-Ladies excepted, the young sir was the father-Loyalty of this nation. My object in making inquiries is not so much about the origin of the plant, which is the object of their art, as about the principles of the formation of that plant. This is known not", - "<|bos|> illness. \n\nPiscatoruddy, July, 1861.\n\nMrs. Crook's daughter actively enjoyed theulsion Rossetti. She, however, did not obtain it in these cases, and indeed, the illness was distressful in the extreme, for Mrs. Lerchy was a very sympathetic person. \n\nShe was quite determined to leave it as a sane man could do.\n\nFor the first time she told me. that she enjoyed it with at least a semester, and intimated how bold the experiment was. The regular exercise played by the firm incessant swinging (sometimes) with the ball dancer, was not a", - "<|bos|>withrepook and sheikhs have never forgetful Siamese readers. Yet in what they have done in stretching or frustrating how he himself accomplished it are apparent to every one. The \n\nSherman illustrated literature and mythology from the Kansas \n\nState Trials; so the lesson of the chapter from Kansas City might have been learned,-from the felicity of men who in his \"Honourable Survey \" have devoted more space to subsidiary and gigantic literature than any other industrial literature. The experiences of \n\nAlfred E. Rogers-the Garrick, with the pieces of Wall hotel at Portland, donning the dress design of Cyrus Box and the campaign of Darwin", - "<|bos|> Kendrician politics, upon his return home notice of the extent and value of the invention he was writing of, proceeded to discuss the problem of Conservative Reform and the difficulties in the path ofReformation, the conditions under which the forces of reaction and progress must take shape, and finally made the statement which is often of the highest authority: It was to be 30 years ago, and then I may say that, for the moment, I was disdainfully and loudly opposed to the change 30 years ago, and feeling most deeply the want lies in modesty, taking every chance to preserve my traditions and to reach my goal. It was", - "<|bos|>Harvard College Library \n\nBOUGHT WITH A INCOME FROM THE BEQUEST OF CHARLES SUMNER, LL. D. \n\nMember of Parliament concerning negroways\n\nOF Harvard College Library \n\nEGATE \u0448\u0435 \n\nMarch 14, 1880\n\nAN U \n\nWH \n\nWILLIAM TROW & SON.\n\nNOV 1929: \n\nSure of the rights of\n\n- Department of Cor \n\nvor \n\nVrol, High Contract\n\nGONE EENT ENL \n\ng Cz1 le lasting v anything upon\n\nPREFACE. \n\nTHE book of this history of events which occurred long since has appeared to me to be one of the greatest calamities of human history. A", - "<|bos|>masters, not of special character. Governor Wise was always very accessible, and had to be both very frequent and repeated. The exclusion of Mr. North, the War Commissioner, from the pulpit, was all that prevented him from gaining some compensation for what he had voluntarily done,-an event which free trade did not long postpone. \n\nPhotograph by fluids. \n\n\"Admiral Porter,\" \"Marguerite\n\nTaylor, tropical gardens, lingam est Octobris, sub-contemporaneous view of Civita \n\nCecenas, Crystal Mountains, Camp Devon, Chatzie, and The Queen Christina stand", - "<|bos|>Army Neglected for beleaguered Petersburg\n\nen Voodone Vortige Eveden Maiefe\n\nannes Nad\u00e9\n\nVrain 1 a Coro 3.5 5.35 Leave was granted \n\n6 15 8.00\n\nFISHERMAN'S RIVELE \n\nby way of effecting a march. It seemed to the wearied men that the fighting was over and the last enemy was advancing. \n\nIn aeffected mood for the first time, they gave gage to Thetis as es Sortige Eulomsdanellov hone Amiaconseins", - "<|bos|>orations. (1) Provided with a library sufficiently standard conferring constructively the-Laurelian and \n\nChantian streams, or the History of \n\nSouth America, and give all instruction in such a subject. Provided with regard to these works the requested attention. \n\n(c) Next to the Agricultural library shall come the Engineering, and a topographical description of the properties occupied by the Western equipment, and afford references for the prices of books in general use by the engineer.\n\nGENERAL NUMBERS. \n\nGENERAL NUMBERS. \n\n6 Various maps and charts illustrating the operations of the British equipped ships of war which fought and captures seventeen countries" - ] -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run2/evals/val_bpb.json b/experiments/think-d12-r11-run2/evals/val_bpb.json deleted file mode 100644 index d5c35476793f6d86232b1da4bfc6ac9077100a95..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.0513915163986667 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run2/run.json b/experiments/think-d12-r11-run2/run.json deleted file mode 100644 index 27b9c4c9937f118ed07f0d9fbbe51ed601156fca..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-r11-run2", - "stage": "base", - "base_experiment_id": "think-d12-r11-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "7f54bf07c5352d97", - "wandb_run_id": "84e31c8e", - "created_at": 1782060945 -} diff --git a/experiments/think-d12-r11-run2/summary.json b/experiments/think-d12-r11-run2/summary.json deleted file mode 100644 index 4409cb7084170c3bb8aa32a889cbdfe5d0d733c0..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "think-d12-r11-run2", - "stage": "base", - "base_experiment_id": "think-d12-r11-run2", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.101818437651454, - "minimum_sampled_val_bpb": 1.101818437651454, - "full_val_bpb": 1.0513915163986667, - "core_metric": 0.06687854185240662, - "centered_results": { - "hellaswag_zeroshot": 0.032463630040486656, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.07002607733011246, - "arc_easy": 0.0740740696589152, - "arc_challenge": -0.0443686048189799, - "copa": 0.059999942779541016, - "commonsense_qa": 0.144144132733345, - "piqa": 0.08487486839294434, - "openbook_qa": -0.015999992688496906, - "lambada_openai": 0.19406171143054962, - "hellaswag": 0.034853617350260414, - "winograd": 0.09890115261077881, - "winogrande": -0.032359957695007324, - "bigbench_dyck_languages": 0.1120000034570694, - "agi_eval_lsat_ar": 0.06521737575531004, - "bigbench_cs_algorithms": 0.3810606002807617, - "bigbench_operators": 0.11428572237491608, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.014569535851478577, - "coqa": 0.08042089641094208, - "boolq": -0.17173665448238973, - "bigbench_language_identification": 0.17436742743249772 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the French nation, and the capital of the French nation, the" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as that of the sun, and the same as that of the moon" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and the day of the week will be the day of the week." - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as cold. \n\nThe latter is the same as the former, and" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, and the planets. \n\n2." - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a red, and I am a little afraid of it. I have been in" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the days of the week, and the number of days of the" - } - ], - "unconditioned_samples": [ - "<|bos|>adjustments. A concern called industrial, they say, is the glory of the Modern Cabinet, and when not mentioned to history, he would be scarcely deserving to be thought a relation.\n\nMr. President, though different from Hume and Lange, we have been met by Dr. Blair and Dr. Hawke, who asked me whether they knew that his father-Ladies excepted, the young sir was the father-Loyalty of this nation. My object in making inquiries is not so much about the origin of the plant, which is the object of their art, as about the principles of the formation of that plant. This is known not", - "<|bos|> illness. \n\nPiscatoruddy, July, 1861.\n\nMrs. Crook's daughter actively enjoyed theulsion Rossetti. She, however, did not obtain it in these cases, and indeed, the illness was distressful in the extreme, for Mrs. Lerchy was a very sympathetic person. \n\nShe was quite determined to leave it as a sane man could do.\n\nFor the first time she told me. that she enjoyed it with at least a semester, and intimated how bold the experiment was. The regular exercise played by the firm incessant swinging (sometimes) with the ball dancer, was not a", - "<|bos|>withrepook and sheikhs have never forgetful Siamese readers. Yet in what they have done in stretching or frustrating how he himself accomplished it are apparent to every one. The \n\nSherman illustrated literature and mythology from the Kansas \n\nState Trials; so the lesson of the chapter from Kansas City might have been learned,-from the felicity of men who in his \"Honourable Survey \" have devoted more space to subsidiary and gigantic literature than any other industrial literature. The experiences of \n\nAlfred E. Rogers-the Garrick, with the pieces of Wall hotel at Portland, donning the dress design of Cyrus Box and the campaign of Darwin", - "<|bos|> Kendrician politics, upon his return home notice of the extent and value of the invention he was writing of, proceeded to discuss the problem of Conservative Reform and the difficulties in the path ofReformation, the conditions under which the forces of reaction and progress must take shape, and finally made the statement which is often of the highest authority: It was to be 30 years ago, and then I may say that, for the moment, I was disdainfully and loudly opposed to the change 30 years ago, and feeling most deeply the want lies in modesty, taking every chance to preserve my traditions and to reach my goal. It was", - "<|bos|>Harvard College Library \n\nBOUGHT WITH A INCOME FROM THE BEQUEST OF CHARLES SUMNER, LL. D. \n\nMember of Parliament concerning negroways\n\nOF Harvard College Library \n\nEGATE \u0448\u0435 \n\nMarch 14, 1880\n\nAN U \n\nWH \n\nWILLIAM TROW & SON.\n\nNOV 1929: \n\nSure of the rights of\n\n- Department of Cor \n\nvor \n\nVrol, High Contract\n\nGONE EENT ENL \n\ng Cz1 le lasting v anything upon\n\nPREFACE. \n\nTHE book of this history of events which occurred long since has appeared to me to be one of the greatest calamities of human history. A", - "<|bos|>masters, not of special character. Governor Wise was always very accessible, and had to be both very frequent and repeated. The exclusion of Mr. North, the War Commissioner, from the pulpit, was all that prevented him from gaining some compensation for what he had voluntarily done,-an event which free trade did not long postpone. \n\nPhotograph by fluids. \n\n\"Admiral Porter,\" \"Marguerite\n\nTaylor, tropical gardens, lingam est Octobris, sub-contemporaneous view of Civita \n\nCecenas, Crystal Mountains, Camp Devon, Chatzie, and The Queen Christina stand", - "<|bos|>Army Neglected for beleaguered Petersburg\n\nen Voodone Vortige Eveden Maiefe\n\nannes Nad\u00e9\n\nVrain 1 a Coro 3.5 5.35 Leave was granted \n\n6 15 8.00\n\nFISHERMAN'S RIVELE \n\nby way of effecting a march. It seemed to the wearied men that the fighting was over and the last enemy was advancing. \n\nIn aeffected mood for the first time, they gave gage to Thetis as es Sortige Eulomsdanellov hone Amiaconseins", - "<|bos|>orations. (1) Provided with a library sufficiently standard conferring constructively the-Laurelian and \n\nChantian streams, or the History of \n\nSouth America, and give all instruction in such a subject. Provided with regard to these works the requested attention. \n\n(c) Next to the Agricultural library shall come the Engineering, and a topographical description of the properties occupied by the Western equipment, and afford references for the prices of books in general use by the engineer.\n\nGENERAL NUMBERS. \n\nGENERAL NUMBERS. \n\n6 Various maps and charts illustrating the operations of the British equipped ships of war which fought and captures seventeen countries" - ], - "training_time_seconds": 6222.97850894928, - "stage_training_flops": 1.0985538644638433e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.0985538644638433e+18, - "config_fingerprint": "7f54bf07c5352d97", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/84e31c8e", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r11-run2", - "dataset_fingerprint": "63a5e6be81591d82", - "tokenizer_fingerprint": "96dd8e502a849d59", - "unique_train_tokens": 1275519304, - "effective_epochs": 0.970873786164196 -} diff --git a/experiments/think-d12-r11-run2/tokenizer/experiment_tokenizer.json b/experiments/think-d12-r11-run2/tokenizer/experiment_tokenizer.json deleted file mode 100644 index ffa040a4514d028ed7fad16e82bee2dccd95ac58..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "think-d12-r11-run2", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1782060961 -} diff --git a/experiments/think-d12-r11-run2/tokenizer/token_bytes.pt b/experiments/think-d12-r11-run2/tokenizer/token_bytes.pt deleted file mode 100644 index 01d1ec4aab9e8a7d205c3b3ffbeb8da0e9a62db1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1 -size 132649 diff --git a/experiments/think-d12-r11-run2/tokenizer/tokenizer.pkl b/experiments/think-d12-r11-run2/tokenizer/tokenizer.pkl deleted file mode 100644 index a17bd392980021628053b95d6425fc556aad527a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run2/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1 -size 404071 diff --git a/experiments/think-d12-r11-run3/base_checkpoints/meta_000500.json b/experiments/think-d12-r11-run3/base_checkpoints/meta_000500.json deleted file mode 100644 index 3e67690938c8373307f1fe4863c4cd2befb1aa49..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 500, - "experiment_id": "think-d12-r11-run3", - "val_bpb": 1.3216701028053668, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run3", - "wandb_run_id": "46b7f9c8", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/base_checkpoints", - "experiment_id": "think-d12-r11-run3", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run3/config.json", - "tokenizer_fingerprint": "85b2a26d1c355860", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 44, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run3", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run3", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 44, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run3", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "dd64eb8e857c4d1a", - "artifact_path": "experiments/think-d12-r11-run3" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run3", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "dd64eb8e857c4d1a" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3216701028053668, - "smooth_train_loss": 3.5803056875068, - "total_training_time": 1297.331782579422, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run3/base_checkpoints/meta_001000.json b/experiments/think-d12-r11-run3/base_checkpoints/meta_001000.json deleted file mode 100644 index 9fd0ffd67fdee5b4ac0cd82e9e07d339caf255c3..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 1000, - "experiment_id": "think-d12-r11-run3", - "val_bpb": 1.2331196818438914, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run3", - "wandb_run_id": "46b7f9c8", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/base_checkpoints", - "experiment_id": "think-d12-r11-run3", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run3/config.json", - "tokenizer_fingerprint": "85b2a26d1c355860", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 44, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run3", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run3", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 44, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run3", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "dd64eb8e857c4d1a", - "artifact_path": "experiments/think-d12-r11-run3" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run3", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "dd64eb8e857c4d1a" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2331196818438914, - "smooth_train_loss": 3.4406616937694783, - "total_training_time": 2622.119250535965, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run3/base_checkpoints/meta_001500.json b/experiments/think-d12-r11-run3/base_checkpoints/meta_001500.json deleted file mode 100644 index 69a7a36cb44c60b3ff0a5feb5e8a62088339f9ee..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 1500, - "experiment_id": "think-d12-r11-run3", - "val_bpb": 1.174251658716475, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run3", - "wandb_run_id": "46b7f9c8", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 1000, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/base_checkpoints", - "experiment_id": "think-d12-r11-run3", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run3/config.json", - "tokenizer_fingerprint": "85b2a26d1c355860", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 44, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run3", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run3", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 44, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run3", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "dd64eb8e857c4d1a", - "artifact_path": "experiments/think-d12-r11-run3" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run3", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "dd64eb8e857c4d1a" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86521538, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86521538 - }, - "loop_state": { - "min_val_bpb": 1.174251658716475, - "smooth_train_loss": 3.270629875998561, - "total_training_time": 4003.970594406128, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run3/base_checkpoints/meta_002000.json b/experiments/think-d12-r11-run3/base_checkpoints/meta_002000.json deleted file mode 100644 index 67da3c8a2070e13b22a251123219ea99709b4d4a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 2000, - "experiment_id": "think-d12-r11-run3", - "val_bpb": 1.1220845787641187, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run3", - "wandb_run_id": "46b7f9c8", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 1000, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/base_checkpoints", - "experiment_id": "think-d12-r11-run3", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run3/config.json", - "tokenizer_fingerprint": "85b2a26d1c355860", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 44, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run3", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run3", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 44, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run3", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "dd64eb8e857c4d1a", - "artifact_path": "experiments/think-d12-r11-run3" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run3", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "dd64eb8e857c4d1a" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48673538, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48673538 - }, - "loop_state": { - "min_val_bpb": 1.1220845787641187, - "smooth_train_loss": 3.242369223248623, - "total_training_time": 5326.6880078315735, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run3/base_checkpoints/meta_002362.json b/experiments/think-d12-r11-run3/base_checkpoints/meta_002362.json deleted file mode 100644 index 724c3b3c081bebb8276ada2c519c53d38a81e343..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 2362, - "experiment_id": "think-d12-r11-run3", - "val_bpb": 1.1016183928831433, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-run3", - "wandb_run_id": "46b7f9c8", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 1000, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11-run3/base_checkpoints", - "experiment_id": "think-d12-r11-run3", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11-run3/config.json", - "tokenizer_fingerprint": "85b2a26d1c355860", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "seed": 44, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11-run3", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run3", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 44, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run3", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "dd64eb8e857c4d1a", - "artifact_path": "experiments/think-d12-r11-run3" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11-run3", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "dd64eb8e857c4d1a" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38471586, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38471586 - }, - "loop_state": { - "min_val_bpb": 1.1016183928831433, - "smooth_train_loss": 3.111287829182217, - "total_training_time": 6288.5381960868835, - "stage_training_flops": 1098553864463843328, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1098553864463843328 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run3/base_checkpoints/model_000500.pt b/experiments/think-d12-r11-run3/base_checkpoints/model_000500.pt deleted file mode 100644 index a7b51a4315d7837ddd8ca931364fefd2d9ea9157..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:281b1d08b67e20c8316754beb155a25f66736a2bb1bc705c782e792b9c5b2e1e -size 792761690 diff --git a/experiments/think-d12-r11-run3/base_checkpoints/model_001000.pt b/experiments/think-d12-r11-run3/base_checkpoints/model_001000.pt deleted file mode 100644 index 8136e7c89e12bc0f4759d5a5b9188dd3aba225bb..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:36c5a9c879666bacd2eff38a8cefd065891833d2d3dfed5b40464c3b71897406 -size 792761690 diff --git a/experiments/think-d12-r11-run3/base_checkpoints/model_001500.pt b/experiments/think-d12-r11-run3/base_checkpoints/model_001500.pt deleted file mode 100644 index 7b46061e2a0cefe3167b8eda8056d6c171952351..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:aaf3c2a3d0451c939f0742a0faff6f246ae25d3b9cebbc62fa6d9e4e27dddac7 -size 792761690 diff --git a/experiments/think-d12-r11-run3/base_checkpoints/model_002000.pt b/experiments/think-d12-r11-run3/base_checkpoints/model_002000.pt deleted file mode 100644 index 32ef78ea87571e6d11dbaca89063341132ad1d6e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:67ebd7cd1447180f6c465b5f7e554f06dbe71855c2b601f3b72a1ac86e759ef4 -size 792761690 diff --git a/experiments/think-d12-r11-run3/base_checkpoints/model_002362.pt b/experiments/think-d12-r11-run3/base_checkpoints/model_002362.pt deleted file mode 100644 index 7ba71b24594223af13a491842e9fa1847b7058ff..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fde26034120b061feb782e40d54d7efec1a8da9bda92f185eb9345749f89db7f -size 792761690 diff --git a/experiments/think-d12-r11-run3/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r11-run3/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 73d97421f2215f2fff8f3748e3fef557f5fd3351..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1b95b914154f1ad2794f0a3aea4ff377d76072ac0c41d442e82db7aea59817e9 -size 1246165357 diff --git a/experiments/think-d12-r11-run3/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r11-run3/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 11ba067553e261025de6af70769bab07d7880495..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d004ede0d800d74ffa6f5f0540afeeba530eb86630bd95c07d88469223365f07 -size 1246165357 diff --git a/experiments/think-d12-r11-run3/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r11-run3/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 0d2631c42bfb103a13cc0c317f69ca53482b13f3..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2de50aba55cc93977007dd0dadcd69dfbf7ccc7ee2709b2fafcb28607714538b -size 1246165357 diff --git a/experiments/think-d12-r11-run3/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r11-run3/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 27b6ed0348bd88bc560cde4e36079b7b4b514d1e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c43386f333717ad577befe466f6aa79408a866500e9a09ff3843f4ac7c484586 -size 1246165357 diff --git a/experiments/think-d12-r11-run3/base_checkpoints/optim_002362_rank0.pt b/experiments/think-d12-r11-run3/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 5c0094ba79390a34776ebe69c83e03d187faa447..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fee73bb716f0dcb81bce2ab65dc7842501e7de6da53f20db7f815f8156749d98 -size 1246165357 diff --git a/experiments/think-d12-r11-run3/config.json b/experiments/think-d12-r11-run3/config.json deleted file mode 100644 index b7ccf5395ba44dcd753f1118d2524a1eb7decb9a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/config.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11-run3", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 44, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-run3", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "dd64eb8e857c4d1a", - "artifact_path": "experiments/think-d12-r11-run3" -} diff --git a/experiments/think-d12-r11-run3/evals/core.json b/experiments/think-d12-r11-run3/evals/core.json deleted file mode 100644 index 60c6a59bd0e6a5b4da21d8397c699a381889cc45..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": 0.06826829817538614, - "core_results": { - "hellaswag_zeroshot": 0.2783310115337372, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.07568524777889252, - "arc_easy": 0.3097642958164215, - "arc_challenge": 0.20307166874408722, - "copa": 0.5099999904632568, - "commonsense_qa": 0.2907452881336212, - "piqa": 0.5435255765914917, - "openbook_qa": 0.24800001084804535, - "lambada_openai": 0.23073936998844147, - "hellaswag": 0.27653852105140686, - "winograd": 0.5384615659713745, - "winogrande": 0.5146014094352722, - "bigbench_dyck_languages": 0.08300000429153442, - "agi_eval_lsat_ar": 0.2695651948451996, - "bigbench_cs_algorithms": 0.42424240708351135, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.01551560964435339, - "coqa": 0.06776900589466095, - "boolq": 0.5620794892311096, - "bigbench_language_identification": 0.2541999816894531 - }, - "centered_results": { - "hellaswag_zeroshot": 0.03777468204498291, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.07568524777889252, - "arc_easy": 0.07968572775522868, - "arc_challenge": -0.06257110834121704, - "copa": 0.019999980926513672, - "commonsense_qa": 0.1134316101670265, - "piqa": 0.0870511531829834, - "openbook_qa": -0.002666652202606201, - "lambada_openai": 0.23073936998844147, - "hellaswag": 0.035384694735209145, - "winograd": 0.07692313194274902, - "winogrande": 0.029202818870544434, - "bigbench_dyck_languages": 0.08300000429153442, - "agi_eval_lsat_ar": 0.08695649355649947, - "bigbench_cs_algorithms": 0.42424240708351135, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.01551560964435339, - "coqa": 0.06776900589466095, - "boolq": -0.15242239676023783, - "bigbench_language_identification": 0.1795379336517636 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run3/evals/samples.json b/experiments/think-d12-r11-run3/evals/samples.json deleted file mode 100644 index ad74ff0fe81dda9009004e4c22819bf878f5f5eb..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the world. \n\nThe capital of the world is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is a gold chain, and the symbol of the gold chain is a gold chain." - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nI am not sure that I shall be able to go to-morrow" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the opposite of cold. \n\nThe former is the more intense, the latter the" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, the stars, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a blue, and I have a blue color. \n\nI have a blue color" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is 13, and 13 = 13, and 13 = 13" - } - ], - "unconditioned_samples": [ - "<|bos|>AL PSALTER PSALTER SCRIPTURE PSALTER CHR CHILDREN \n\nChristenome for Modern Sinners Psamates Sallarememorative Psalms Biblioteca \n\nDalmatique Italiano\n\nEaster exegetica Elisha Scotti Netherlandsnio Italiano \n\nDante Bert Raphael Exposition Cant Cant Cant Cant Cant Cant Cant Cant cant\n\nBible Jewish D-Lo spelling Cant Cant \n\nInitiatory Levant Horne\n\nItaliano-Anglo Jupe\n\nFolklore Spanish Olino Florinda In festo \n\nItaliano-Romano Guglielmoi \n\nGui Ser vocali Ministeri Burton", - "<|bos|> salary of the past year, including French and German assistance. The prospects of the business are very gloomy in this country. Rossetti is living, as a manager, in perfect peace. Often, it is said, he sets his empty purse at 73 per cent. It is probable that a very large amount of this is withdrawn every year. But it is probable also that the experience of the last two years is very different. He has been connected with at least one newspaper, having never been in any enterprise except the one that regular news service requires. One incessant fight he made with an old woman who was drugging and disguising herself", - "<|bos|> us Free men and she who could never excuse forgetful Siamese. Well, in the last grand war such men had nothing to lose. Thank you, for your rain, and say 'Good-bye.' I'll all this long. \n\nWith kind Lady Cassoit. \n\nYours very affectionately, \n\nSQUIRE SHANGHANK. \n\nL. SHERgee.\n\n'You were got so devotedly tied up by feeling to this morning that you struggled against it long with all your might when it was taken from me. But, as soon as you had won your way to your old friends and dear home, then I had", - "<|bos|> Kendenthal, Brig.-Gen. \n\nWaller, Mutiny Washington, 9 U.S. 451.\n\nMaryland loss at Manila.-Send prompt orders in case there is any error, the business is completed in twenty days.\n\nAmerican forces at Geba.---Actinon.-S. of the enemy at Geba. C.B. from active duties, natives, tude; valor (Georgia men in hospital), and privates, militia, 3,000 tons War Department. removed shipping lies to the colonies taking contraband of war, removing twelve hundred stone from one stockading", - "<|bos|> HOUSE OF THE LION RIOTS. \n\nAgate.] One who is anxious to get a glimpse into the future of My Majesty in this world, and out of the midst of error, far from denying himself marriage, treats it one who has hitherto trusted himself to carry on the war with difficulty and success: unincorporated, not, as some think, out of discouragement, but because he has shown himself so willing to enter on higher ground; often, not, perhaps, without some little reflection for his own sake, but without question after long and fatiguing deliberation; but, when earnest, unwearied and useful labour rises up", - "<|bos|>1690, 1 August. Governor Nathaniel Claffel. \n\nVan Isle.\n\nU. S. Army. \n\n1 June. General Cooper. \n\n1 June. Colonel Theodore Astley. \n\n1 July. Commander Joseph Palmer.\n\nEntom., Joshua Billings, John Hill, \n\n8 June. Captain John Daniels \n\nPhoebe Talbotton \n\n\"Admiral Porter,\" \"Marguerite\n\nNewburne \n\ntwo or three names for Oct. 14th. \n\nBertie Mills\n\nEmery Park\n\nLaugwanton \n\nCapt. William Ludlow \n\nHezekiah Meyers \n\nMosby", - "<|bos|>Army in the field and defends itself by an effective briinjade. He did not resort to threats, nor to use a poniard; but he only levied dreadfully upon the French who were marching up and down the Army.\n\n\"How should it be practicable now?\" said theLOWERS, while the Army was effecting its march. \"The army should be kept intact. The furnishing of the necessary subsistence should be strictly enforced. Thee should have these river-banks filled up, and garrisons The Executive has orders to inflict extreme corrup-\n\nviction soon on the enemy, because they are", - "<|bos|>the eightieth year of his age.\n\nI had never been conferring, like the paleface VIII. of the Conquests, on the History of \n\nSouth America, and give all the name ever given this island to Penrith.' \n\nLectures to the Greeks. \n\n[See preceding list.] 119-113, and a vaulted great hall.- \n\nApart from the three triumphs above described, there are three books in the Museum that are universally known to the present generation of students. The last has the date \n\n1184.'-Journal of the Institute of Arts, Sciences and Arts, No. 725" - ] -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run3/evals/val_bpb.json b/experiments/think-d12-r11-run3/evals/val_bpb.json deleted file mode 100644 index 0f4a6908a0bafe7e43f38a10d9237eaf5ef603f4..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.0504735756248413 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11-run3/run.json b/experiments/think-d12-r11-run3/run.json deleted file mode 100644 index 557535ce6a279f6a53b41e5a65e1e399c2e31f2c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-r11-run3", - "stage": "base", - "base_experiment_id": "think-d12-r11-run3", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "dd64eb8e857c4d1a", - "wandb_run_id": "46b7f9c8", - "created_at": 1782133395 -} diff --git a/experiments/think-d12-r11-run3/summary.json b/experiments/think-d12-r11-run3/summary.json deleted file mode 100644 index 5aaf8e216cbb8f830bc9573da6a414fd1883a760..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "think-d12-r11-run3", - "stage": "base", - "base_experiment_id": "think-d12-r11-run3", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.1016183928831433, - "minimum_sampled_val_bpb": 1.1016183928831433, - "full_val_bpb": 1.0504735756248413, - "core_metric": 0.06826829817538614, - "centered_results": { - "hellaswag_zeroshot": 0.03777468204498291, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.07568524777889252, - "arc_easy": 0.07968572775522868, - "arc_challenge": -0.06257110834121704, - "copa": 0.019999980926513672, - "commonsense_qa": 0.1134316101670265, - "piqa": 0.0870511531829834, - "openbook_qa": -0.002666652202606201, - "lambada_openai": 0.23073936998844147, - "hellaswag": 0.035384694735209145, - "winograd": 0.07692313194274902, - "winogrande": 0.029202818870544434, - "bigbench_dyck_languages": 0.08300000429153442, - "agi_eval_lsat_ar": 0.08695649355649947, - "bigbench_cs_algorithms": 0.42424240708351135, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.01551560964435339, - "coqa": 0.06776900589466095, - "boolq": -0.15242239676023783, - "bigbench_language_identification": 0.1795379336517636 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the world. \n\nThe capital of the world is the capital of" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is a gold chain, and the symbol of the gold chain is a gold chain." - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nI am not sure that I shall be able to go to-morrow" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the opposite of cold. \n\nThe former is the more intense, the latter the" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, the stars, the sun, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a blue, and I have a blue color. \n\nI have a blue color" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is 13, and 13 = 13, and 13 = 13" - } - ], - "unconditioned_samples": [ - "<|bos|>AL PSALTER PSALTER SCRIPTURE PSALTER CHR CHILDREN \n\nChristenome for Modern Sinners Psamates Sallarememorative Psalms Biblioteca \n\nDalmatique Italiano\n\nEaster exegetica Elisha Scotti Netherlandsnio Italiano \n\nDante Bert Raphael Exposition Cant Cant Cant Cant Cant Cant Cant Cant cant\n\nBible Jewish D-Lo spelling Cant Cant \n\nInitiatory Levant Horne\n\nItaliano-Anglo Jupe\n\nFolklore Spanish Olino Florinda In festo \n\nItaliano-Romano Guglielmoi \n\nGui Ser vocali Ministeri Burton", - "<|bos|> salary of the past year, including French and German assistance. The prospects of the business are very gloomy in this country. Rossetti is living, as a manager, in perfect peace. Often, it is said, he sets his empty purse at 73 per cent. It is probable that a very large amount of this is withdrawn every year. But it is probable also that the experience of the last two years is very different. He has been connected with at least one newspaper, having never been in any enterprise except the one that regular news service requires. One incessant fight he made with an old woman who was drugging and disguising herself", - "<|bos|> us Free men and she who could never excuse forgetful Siamese. Well, in the last grand war such men had nothing to lose. Thank you, for your rain, and say 'Good-bye.' I'll all this long. \n\nWith kind Lady Cassoit. \n\nYours very affectionately, \n\nSQUIRE SHANGHANK. \n\nL. SHERgee.\n\n'You were got so devotedly tied up by feeling to this morning that you struggled against it long with all your might when it was taken from me. But, as soon as you had won your way to your old friends and dear home, then I had", - "<|bos|> Kendenthal, Brig.-Gen. \n\nWaller, Mutiny Washington, 9 U.S. 451.\n\nMaryland loss at Manila.-Send prompt orders in case there is any error, the business is completed in twenty days.\n\nAmerican forces at Geba.---Actinon.-S. of the enemy at Geba. C.B. from active duties, natives, tude; valor (Georgia men in hospital), and privates, militia, 3,000 tons War Department. removed shipping lies to the colonies taking contraband of war, removing twelve hundred stone from one stockading", - "<|bos|> HOUSE OF THE LION RIOTS. \n\nAgate.] One who is anxious to get a glimpse into the future of My Majesty in this world, and out of the midst of error, far from denying himself marriage, treats it one who has hitherto trusted himself to carry on the war with difficulty and success: unincorporated, not, as some think, out of discouragement, but because he has shown himself so willing to enter on higher ground; often, not, perhaps, without some little reflection for his own sake, but without question after long and fatiguing deliberation; but, when earnest, unwearied and useful labour rises up", - "<|bos|>1690, 1 August. Governor Nathaniel Claffel. \n\nVan Isle.\n\nU. S. Army. \n\n1 June. General Cooper. \n\n1 June. Colonel Theodore Astley. \n\n1 July. Commander Joseph Palmer.\n\nEntom., Joshua Billings, John Hill, \n\n8 June. Captain John Daniels \n\nPhoebe Talbotton \n\n\"Admiral Porter,\" \"Marguerite\n\nNewburne \n\ntwo or three names for Oct. 14th. \n\nBertie Mills\n\nEmery Park\n\nLaugwanton \n\nCapt. William Ludlow \n\nHezekiah Meyers \n\nMosby", - "<|bos|>Army in the field and defends itself by an effective briinjade. He did not resort to threats, nor to use a poniard; but he only levied dreadfully upon the French who were marching up and down the Army.\n\n\"How should it be practicable now?\" said theLOWERS, while the Army was effecting its march. \"The army should be kept intact. The furnishing of the necessary subsistence should be strictly enforced. Thee should have these river-banks filled up, and garrisons The Executive has orders to inflict extreme corrup-\n\nviction soon on the enemy, because they are", - "<|bos|>the eightieth year of his age.\n\nI had never been conferring, like the paleface VIII. of the Conquests, on the History of \n\nSouth America, and give all the name ever given this island to Penrith.' \n\nLectures to the Greeks. \n\n[See preceding list.] 119-113, and a vaulted great hall.- \n\nApart from the three triumphs above described, there are three books in the Museum that are universally known to the present generation of students. The last has the date \n\n1184.'-Journal of the Institute of Arts, Sciences and Arts, No. 725" - ], - "training_time_seconds": 6288.5381960868835, - "stage_training_flops": 1.0985538644638433e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.0985538644638433e+18, - "config_fingerprint": "dd64eb8e857c4d1a", - "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/46b7f9c8", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r11-run3", - "dataset_fingerprint": "63a5e6be81591d82", - "tokenizer_fingerprint": "85b2a26d1c355860", - "unique_train_tokens": 1275519304, - "effective_epochs": 0.970873786164196 -} diff --git a/experiments/think-d12-r11-run3/tokenizer/experiment_tokenizer.json b/experiments/think-d12-r11-run3/tokenizer/experiment_tokenizer.json deleted file mode 100644 index fc70eca95e86a79db88161e6f68918d1b53cbe91..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "think-d12-r11-run3", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1782133420 -} diff --git a/experiments/think-d12-r11-run3/tokenizer/token_bytes.pt b/experiments/think-d12-r11-run3/tokenizer/token_bytes.pt deleted file mode 100644 index 01d1ec4aab9e8a7d205c3b3ffbeb8da0e9a62db1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1 -size 132649 diff --git a/experiments/think-d12-r11-run3/tokenizer/tokenizer.pkl b/experiments/think-d12-r11-run3/tokenizer/tokenizer.pkl deleted file mode 100644 index a17bd392980021628053b95d6425fc556aad527a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11-run3/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1 -size 404071 diff --git a/experiments/think-d12-r11.25-ctx4096-sssl/config.json b/experiments/think-d12-r11.25-ctx4096-sssl/config.json deleted file mode 100644 index 768784a0486010588865a9c7e299d1106131eefc..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096-sssl/config.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx4096-sssl", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "SSSL", - "device_batch_size": 8, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx4096-sssl", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx4096", - "sssl" - ] - }, - "config_fingerprint": "6e9c75a327904b86", - "artifact_path": "experiments/think-d12-r11.25-ctx4096-sssl" -} diff --git a/experiments/think-d12-r11.25-ctx4096-sssl/run.json b/experiments/think-d12-r11.25-ctx4096-sssl/run.json deleted file mode 100644 index 4ba9f266da85ae23b7ef452be2ffcbd0b4a1b58d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096-sssl/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-r11.25-ctx4096-sssl", - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx4096-sssl", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6e9c75a327904b86", - "wandb_run_id": "d925b571", - "created_at": 1784302194 -} diff --git a/experiments/think-d12-r11.25-ctx4096-sssl/tokenizer/experiment_tokenizer.json b/experiments/think-d12-r11.25-ctx4096-sssl/tokenizer/experiment_tokenizer.json deleted file mode 100644 index b492eed0c58653d83091c8723e1267fde869e135..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096-sssl/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "think-d12-r11.25-ctx4096-sssl", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1784302214 -} diff --git a/experiments/think-d12-r11.25-ctx4096-sssl/tokenizer/token_bytes.pt b/experiments/think-d12-r11.25-ctx4096-sssl/tokenizer/token_bytes.pt deleted file mode 100644 index 52b8c6160971208bdb8a09f45e79882217012ba5..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096-sssl/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:89a69da6264286005b0e65be12a1afc9bc44936d535527f347d22ab74790076c -size 132649 diff --git a/experiments/think-d12-r11.25-ctx4096-sssl/tokenizer/tokenizer.pkl b/experiments/think-d12-r11.25-ctx4096-sssl/tokenizer/tokenizer.pkl deleted file mode 100644 index 3d323d48ca5e52338e432ef2a6ed24f12b15c4e8..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096-sssl/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2df5bd08e7e921a0404abe4642907707baa66c6840e74a90aa16d16b268bd5d1 -size 404037 diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_000500.json b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_000500.json deleted file mode 100644 index 621a745b56f619d244c4cebe618fab1a46780a38..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "think-d12-r11.25-ctx4096", - "val_bpb": 1.3085028476600495, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-ctx4096", - "wandb_run_id": "85989fb4", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25,ctx4096", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/base_checkpoints", - "experiment_id": "think-d12-r11.25-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/config.json", - "tokenizer_fingerprint": "7ad989ceea17794b", - "git_commit_sha": "128fb5d3c7f5e7122b6eb4390cf207e7e986d9dd", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11.25-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx4096", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx4096" - ] - }, - "config_fingerprint": "a1e3ff2a651b87ea", - "artifact_path": "experiments/think-d12-r11.25-ctx4096" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a1e3ff2a651b87ea" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3085028476600495, - "smooth_train_loss": 3.5543985215236638, - "total_training_time": 1486.219638824463, - "stage_training_flops": 291921016651776000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 291921016651776000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_001000.json b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_001000.json deleted file mode 100644 index 8ff79e207d7bf5cf702bdb5fd2f9761ea2a360fb..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "think-d12-r11.25-ctx4096", - "val_bpb": 1.209916555955143, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-ctx4096", - "wandb_run_id": "85989fb4", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25,ctx4096", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 500, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/base_checkpoints", - "experiment_id": "think-d12-r11.25-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/config.json", - "tokenizer_fingerprint": "7ad989ceea17794b", - "git_commit_sha": "128fb5d3c7f5e7122b6eb4390cf207e7e986d9dd", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11.25-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx4096", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx4096" - ] - }, - "config_fingerprint": "a1e3ff2a651b87ea", - "artifact_path": "experiments/think-d12-r11.25-ctx4096" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a1e3ff2a651b87ea" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24369538, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24369538 - }, - "loop_state": { - "min_val_bpb": 1.209916555955143, - "smooth_train_loss": 3.4094984750904938, - "total_training_time": 3055.330541372299, - "stage_training_flops": 583842033303552000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 583842033303552000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_001500.json b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_001500.json deleted file mode 100644 index ddbfd68db055843968eb36669e5c791ed83bd0e4..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "think-d12-r11.25-ctx4096", - "val_bpb": 1.1509853231334473, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-ctx4096", - "wandb_run_id": "85989fb4", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25,ctx4096", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 500, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/base_checkpoints", - "experiment_id": "think-d12-r11.25-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/config.json", - "tokenizer_fingerprint": "7ad989ceea17794b", - "git_commit_sha": "128fb5d3c7f5e7122b6eb4390cf207e7e986d9dd", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11.25-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx4096", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx4096" - ] - }, - "config_fingerprint": "a1e3ff2a651b87ea", - "artifact_path": "experiments/think-d12-r11.25-ctx4096" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a1e3ff2a651b87ea" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86521538, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86521538 - }, - "loop_state": { - "min_val_bpb": 1.1509853231334473, - "smooth_train_loss": 3.2151429122859003, - "total_training_time": 4564.625297546387, - "stage_training_flops": 875763049955328000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 875763049955328000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_002000.json b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_002000.json deleted file mode 100644 index 102758e5e738a387460507ad0661b7ef5c899772..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "think-d12-r11.25-ctx4096", - "val_bpb": 1.099780461697027, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-ctx4096", - "wandb_run_id": "85989fb4", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25,ctx4096", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 500, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/base_checkpoints", - "experiment_id": "think-d12-r11.25-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/config.json", - "tokenizer_fingerprint": "7ad989ceea17794b", - "git_commit_sha": "128fb5d3c7f5e7122b6eb4390cf207e7e986d9dd", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11.25-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx4096", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx4096" - ] - }, - "config_fingerprint": "a1e3ff2a651b87ea", - "artifact_path": "experiments/think-d12-r11.25-ctx4096" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a1e3ff2a651b87ea" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48673538, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48673538 - }, - "loop_state": { - "min_val_bpb": 1.099780461697027, - "smooth_train_loss": 3.1695909607341712, - "total_training_time": 6075.22934627533, - "stage_training_flops": 1167684066607104000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1167684066607104000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_002362.json b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_002362.json deleted file mode 100644 index 65a6a0092c2c3e6b33a4c88fe36c0606a1454473..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "think-d12-r11.25-ctx4096", - "val_bpb": 1.0795015991364572, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-ctx4096", - "wandb_run_id": "85989fb4", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25,ctx4096", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 500, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/base_checkpoints", - "experiment_id": "think-d12-r11.25-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx4096/config.json", - "tokenizer_fingerprint": "7ad989ceea17794b", - "git_commit_sha": "128fb5d3c7f5e7122b6eb4390cf207e7e986d9dd", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11.25-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx4096", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx4096" - ] - }, - "config_fingerprint": "a1e3ff2a651b87ea", - "artifact_path": "experiments/think-d12-r11.25-ctx4096" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a1e3ff2a651b87ea" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38471586, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38471586 - }, - "loop_state": { - "min_val_bpb": 1.0795015991364572, - "smooth_train_loss": 3.0619805740794512, - "total_training_time": 7167.931929588318, - "stage_training_flops": 1379034882662989824, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1379034882662989824 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_000500.pt b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_000500.pt deleted file mode 100644 index 4c3181f54124f5ba051f2c3b4dc0f973adb5ceb2..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:eae964a0bb2c1375c3b4451c8aa5570663349d91ee3b321c472df0d3a4660219 -size 792761690 diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_001000.pt b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_001000.pt deleted file mode 100644 index 571aa5a944ca0b8b5889c94a67706257db85131e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a66e11e92bed0abf8702c6553dfaed10a232a99c9b6e9d90303d5847430adc1d -size 792761690 diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_001500.pt b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_001500.pt deleted file mode 100644 index cbd23725cea6e13eb84db458267ace1699d65d01..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:07b77aef3943e2b0cb0434580c50489846faaefcf579df38c79139f1f1b16093 -size 792761690 diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_002000.pt b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_002000.pt deleted file mode 100644 index 7e23fb9eaae0e92d29d0472571dd2fff20107b47..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c26eb6938cd1b71b2f3b66d7287f6bc768184242413062ece92523a88694d434 -size 792761690 diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_002362.pt b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_002362.pt deleted file mode 100644 index 07ea2f582c28be0bebb4d785113f53a6af83b1dc..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5925ab0fb833c557d0252f792cc84f5b538514defc131f2dae5eb3ebcd038705 -size 792761690 diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index f1e7adecb5041fd09f6e1a50f21df76a741ccc9c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:df415f6eb227a7508933cf3e56d35b580e1c4fc294b213ab3913340595f5cfdc -size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 2a421f2700606a3cedb0c4b930ea2b10ef13ed3a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3039b07883f2b42253768f446c4c007427398b7b019300365ea7be9e0fc42ad5 -size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 793b9df065ed3fe1868036306fcd30952919c8f4..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2407aa8a07d26bd2f67674ecd6849bea4ba0da3577680f8b1fdba446ac0e1fe8 -size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 5ad8110c17ac5e0fde78cad1a9f90d34136edec6..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2d22fa6ba43269e9464c107e377955f93e7032ca892a20961f6dae3ab8800b2f -size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_002362_rank0.pt b/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 1bafee36bb702704c1b5ab8e22812dcb4369adbc..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:59ebb8f16f64fdc23ef0136ea29fd602db7eb76ed53c0b4b36167d1b65ae06dc -size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx4096/config.json b/experiments/think-d12-r11.25-ctx4096/config.json deleted file mode 100644 index c0b1edb621d47cf9c5eb14744f0c1d0db7ffa254..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/config.json +++ /dev/null @@ -1,57 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx4096", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx4096" - ] - }, - "config_fingerprint": "a1e3ff2a651b87ea", - "artifact_path": "experiments/think-d12-r11.25-ctx4096" -} diff --git a/experiments/think-d12-r11.25-ctx4096/evals/core.json b/experiments/think-d12-r11.25-ctx4096/evals/core.json deleted file mode 100644 index 18bd17537653676d6499795eafc1facad7c7104d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": 0.06440192842734686, - "core_results": { - "hellaswag_zeroshot": 0.2777335047721863, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.09128487855195999, - "arc_easy": 0.31186866760253906, - "arc_challenge": 0.21160408854484558, - "copa": 0.5099999904632568, - "commonsense_qa": 0.31449630856513977, - "piqa": 0.5369967222213745, - "openbook_qa": 0.24800001084804535, - "lambada_openai": 0.23015718162059784, - "hellaswag": 0.2796255648136139, - "winograd": 0.553113579750061, - "winogrande": 0.5106551051139832, - "bigbench_dyck_languages": 0.10200000554323196, - "agi_eval_lsat_ar": 0.2956521511077881, - "bigbench_cs_algorithms": 0.40303027629852295, - "bigbench_operators": 0.0714285746216774, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.02308420091867447, - "coqa": 0.06664161384105682, - "boolq": 0.49082568287849426, - "bigbench_language_identification": 0.25360000133514404 - }, - "centered_results": { - "hellaswag_zeroshot": 0.03697800636291504, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.09128487855195999, - "arc_easy": 0.08249155680338542, - "arc_challenge": -0.05119454860687256, - "copa": 0.019999980926513672, - "commonsense_qa": 0.14312038570642469, - "piqa": 0.07399344444274902, - "openbook_qa": -0.002666652202606201, - "lambada_openai": 0.23015718162059784, - "hellaswag": 0.03950075308481852, - "winograd": 0.10622715950012207, - "winogrande": 0.02131021022796631, - "bigbench_dyck_languages": 0.10200000554323196, - "agi_eval_lsat_ar": 0.1195651888847351, - "bigbench_cs_algorithms": 0.40303027629852295, - "bigbench_operators": 0.0714285746216774, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.02308420091867447, - "coqa": 0.06664161384105682, - "boolq": -0.3399324134776467, - "bigbench_language_identification": 0.17887788925758422 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx4096/evals/samples.json b/experiments/think-d12-r11.25-ctx4096/evals/samples.json deleted file mode 100644 index bc023780539cb6a95be1c927a096d903f9c1dc3c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is not yet exhausted, and the French are not yet in the field. \n\nThe" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the gold of the earth. \n\nThe gold of the earth is" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be the day of the Lord's coming. \n\nThe Lord will be with you," - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the opposite of cold. \n\nThe former is the opposite of cold, and the" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, which is the centre of the solar system, and" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the color of the skin, and the color of the skin. \n\nThe color" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the same as 13, and the same as 13, and the same" - } - ], - "unconditioned_samples": [ - "<|bos|>ALIENO. A term applied to some portion of Alcoholism which is for local use, is used not only to signify animal foods or grains for eating, but also to all grains between or under the different degrees of nutrition; or, for local use, to all meat between the quality of the food itself and the character of its flavor (American Journal of Medicine, XXI, p. 5). \n\nAlcoholic poisons are habitually given out with so much care that sometimes they are exhausted in the form of alcoholism; or, if this should not be the case, were it so, Alcohol would", - "<|bos|>orted. \n\nPiscacy becomes a science, or a profession.\n\nAnd one considers the fact that different animals, birds, and dragon-flies do swim across this continent. \n\nStill, however, we should have to consider the tails of the wings of the \n\nInsects, if we would know how to unlock their wings. That some of the wings insects swim across such a space. \n\nTHE PLACE OF THE-PLAN OF THE insect. \n\nIt was the desire of the average botanist to converse along the coast of the \n\nSoutheast coast with these four duskyton insects, while the", - "<|bos|>ROUBSTANTI, who rendered justly such service to Si- \n\nMure in understanding her granddaughter, stretching from her little nose downwards towards it to the exclusion of his voice, thus speaks the feeble lines of La \n\nFriente: \n\nBut Lady Cassoit. \n\nDon Diego! \n\nMichel of Carnival! \n\nMy father! \n\nLa M'Enfer. \n\nHot-spur of subsidiary! \n\nI serv'd madame Alphonso, \n\nThat when she heard-the nasty child,\n\nAnd was in hotel at home, don't know her, Miss SCALE to schoolwoman. My", - "<|bos|>HENRY MART 140.6. \n\nWaller Lectures on the Science of Religious Belief - Vol. 4. pp. 213-407. \n\nHenry Martineau's Notes on Religious Belief. Vol. 1. Political Essays. \n\npt. 2. pp. 450-714. He has noticed various essays to popularize certain philosophy of Error, chiefly controversial, compassionate, and (as it would be well that these should not be looked at 3 very badly, and Father Matheo has suggested lies to modesty in taking a lead to preserve us from the tremendous reach of such false guides.", - "<|bos|>Harvard College Library \n\nBOUGHT WITH \n\nTHE INCOME FROM THE BEQUEST OF MRS. HARRIET J. BRADBURY \n\n(1 My Last in Berkshire\n\nCountess of Salisbury. \n\n\u0448\u0435 \n\nMarch \n\n1443-44 \n\nMarch 1439-42\n\nMarch \n\nE.\n\nMarch \n\nHITE: - \n\nOf the \n\nMarch \n\n-March\n\nMarch \n\nApril April \n\nevrolment of his Fellowships' design could no more be concealed than that it was intended to tell his friends that which was not long ago published, and therefore after long and not-too uniform a course, he was resolved to leave the real scheme out", - "<|bos|>The commerce of the world is in the world the pelagic war. \n\nUntil within the memory of man, this task has never been assigned to a privileged class. Astutians and all peoples have attempted to advance some modest policy, and have entered into a harmonious co-operation in order to conduct an honourable life. \n\nPharaoh has not spared his fatherland the repetition of arts, that marred his inheritance, and has left plenty for his architects and his sub-contractor. \n\nWho may assert here that in the event of attempts to modify these methods he has not been mistaken, is without The Queen es death and", - "<|bos|>Army of the Cumberland and Cumberland Rivers, Vol. V, No. 2.\n\n-training of French troops on the Cumberland and towards Pulaski between 1850 and 1855. \n\nmarched or red with a. \n\n6 66 \n\n8. Soberly County, five miles from Lancaster, Anthony County,\n\nPassy County; Adams County, two miles from Pembroke and Fayette County; or Pakenham County, one miles from Augusta, York County, Bodega County; Meaningley, half a mile north The Confede esse moved by Way of the Cumberland to the north hone.\n\nDetroit. \n\n", - "<|bos|>the eightieth year of his age.\n\nI had never been conferring with men on the principal routes to see the prospect, and, on the other hand, when it was discovered, the probability became ever less that the enemy were countless in number. There was not enough intelligence in collecting at his house to prevent his asking aid. Moreover, he would not have great confidence in the number of his soldiers, and would not require reinforcements. Human nature was too much enamoured to allow men to render him service. \n\nWith the handful of soldiers on this side, and the army in apparently ill condition, the army was utterly destroyed." - ] -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx4096/evals/val_bpb.json b/experiments/think-d12-r11.25-ctx4096/evals/val_bpb.json deleted file mode 100644 index d08fde68a81cb60a38b7cc13d7dab2e4f1bebffd..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/evals/val_bpb.json +++ /dev/null @@ -1,94 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val_per_position": [ - { - "start": 0, - "end": 256, - "bpb": 1.130142936359072 - }, - { - "start": 256, - "end": 512, - "bpb": 1.065738621390211 - }, - { - "start": 512, - "end": 768, - "bpb": 1.0495748692031812 - }, - { - "start": 768, - "end": 1024, - "bpb": 1.0430983472198936 - }, - { - "start": 1024, - "end": 1280, - "bpb": 1.0366271994096037 - }, - { - "start": 1280, - "end": 1536, - "bpb": 1.0297924628551816 - }, - { - "start": 1536, - "end": 1792, - "bpb": 1.0281174126636314 - }, - { - "start": 1792, - "end": 2048, - "bpb": 1.0244993749695677 - }, - { - "start": 2048, - "end": 2304, - "bpb": 1.020344951749226 - }, - { - "start": 2304, - "end": 2560, - "bpb": 1.017900061022826 - }, - { - "start": 2560, - "end": 2816, - "bpb": 1.017925884634094 - }, - { - "start": 2816, - "end": 3072, - "bpb": 1.0206878528813161 - }, - { - "start": 3072, - "end": 3328, - "bpb": 1.0171357038345088 - }, - { - "start": 3328, - "end": 3584, - "bpb": 1.0150316846511822 - }, - { - "start": 3584, - "end": 3840, - "bpb": 1.0124244006389753 - }, - { - "start": 3840, - "end": 4096, - "bpb": 1.0101240407476038 - } - ], - "val": 1.0336899522729732 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx4096/run.json b/experiments/think-d12-r11.25-ctx4096/run.json deleted file mode 100644 index d454763adc550fdad06c22c0a57fd590d19bf077..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-r11.25-ctx4096", - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "a1e3ff2a651b87ea", - "wandb_run_id": "85989fb4", - "created_at": 1783526801 -} diff --git a/experiments/think-d12-r11.25-ctx4096/summary.json b/experiments/think-d12-r11.25-ctx4096/summary.json deleted file mode 100644 index 307f246e92ce97c157179f85ebebb5595e57bc57..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/summary.json +++ /dev/null @@ -1,69 +0,0 @@ -{ - "experiment_id": "think-d12-r11.25-ctx4096", - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.0795015991364572, - "minimum_sampled_val_bpb": 1.0795015991364572, - "full_val_bpb": 1.0336899522729732, - "core_metric": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is not yet exhausted, and the French are not yet in the field. \n\nThe" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the gold of the earth. \n\nThe gold of the earth is" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be the day of the Lord's coming. \n\nThe Lord will be with you," - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the opposite of cold. \n\nThe former is the opposite of cold, and the" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, which is the centre of the solar system, and" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the color of the skin, and the color of the skin. \n\nThe color" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the same as 13, and the same as 13, and the same" - } - ], - "unconditioned_samples": [ - "<|bos|>ALIENO. A term applied to some portion of Alcoholism which is for local use, is used not only to signify animal foods or grains for eating, but also to all grains between or under the different degrees of nutrition; or, for local use, to all meat between the quality of the food itself and the character of its flavor (American Journal of Medicine, XXI, p. 5). \n\nAlcoholic poisons are habitually given out with so much care that sometimes they are exhausted in the form of alcoholism; or, if this should not be the case, were it so, Alcohol would", - "<|bos|>orted. \n\nPiscacy becomes a science, or a profession.\n\nAnd one considers the fact that different animals, birds, and dragon-flies do swim across this continent. \n\nStill, however, we should have to consider the tails of the wings of the \n\nInsects, if we would know how to unlock their wings. That some of the wings insects swim across such a space. \n\nTHE PLACE OF THE-PLAN OF THE insect. \n\nIt was the desire of the average botanist to converse along the coast of the \n\nSoutheast coast with these four duskyton insects, while the", - "<|bos|>ROUBSTANTI, who rendered justly such service to Si- \n\nMure in understanding her granddaughter, stretching from her little nose downwards towards it to the exclusion of his voice, thus speaks the feeble lines of La \n\nFriente: \n\nBut Lady Cassoit. \n\nDon Diego! \n\nMichel of Carnival! \n\nMy father! \n\nLa M'Enfer. \n\nHot-spur of subsidiary! \n\nI serv'd madame Alphonso, \n\nThat when she heard-the nasty child,\n\nAnd was in hotel at home, don't know her, Miss SCALE to schoolwoman. My", - "<|bos|>HENRY MART 140.6. \n\nWaller Lectures on the Science of Religious Belief - Vol. 4. pp. 213-407. \n\nHenry Martineau's Notes on Religious Belief. Vol. 1. Political Essays. \n\npt. 2. pp. 450-714. He has noticed various essays to popularize certain philosophy of Error, chiefly controversial, compassionate, and (as it would be well that these should not be looked at 3 very badly, and Father Matheo has suggested lies to modesty in taking a lead to preserve us from the tremendous reach of such false guides.", - "<|bos|>Harvard College Library \n\nBOUGHT WITH \n\nTHE INCOME FROM THE BEQUEST OF MRS. HARRIET J. BRADBURY \n\n(1 My Last in Berkshire\n\nCountess of Salisbury. \n\n\u0448\u0435 \n\nMarch \n\n1443-44 \n\nMarch 1439-42\n\nMarch \n\nE.\n\nMarch \n\nHITE: - \n\nOf the \n\nMarch \n\n-March\n\nMarch \n\nApril April \n\nevrolment of his Fellowships' design could no more be concealed than that it was intended to tell his friends that which was not long ago published, and therefore after long and not-too uniform a course, he was resolved to leave the real scheme out", - "<|bos|>The commerce of the world is in the world the pelagic war. \n\nUntil within the memory of man, this task has never been assigned to a privileged class. Astutians and all peoples have attempted to advance some modest policy, and have entered into a harmonious co-operation in order to conduct an honourable life. \n\nPharaoh has not spared his fatherland the repetition of arts, that marred his inheritance, and has left plenty for his architects and his sub-contractor. \n\nWho may assert here that in the event of attempts to modify these methods he has not been mistaken, is without The Queen es death and", - "<|bos|>Army of the Cumberland and Cumberland Rivers, Vol. V, No. 2.\n\n-training of French troops on the Cumberland and towards Pulaski between 1850 and 1855. \n\nmarched or red with a. \n\n6 66 \n\n8. Soberly County, five miles from Lancaster, Anthony County,\n\nPassy County; Adams County, two miles from Pembroke and Fayette County; or Pakenham County, one miles from Augusta, York County, Bodega County; Meaningley, half a mile north The Confede esse moved by Way of the Cumberland to the north hone.\n\nDetroit. \n\n", - "<|bos|>the eightieth year of his age.\n\nI had never been conferring with men on the principal routes to see the prospect, and, on the other hand, when it was discovered, the probability became ever less that the enemy were countless in number. There was not enough intelligence in collecting at his house to prevent his asking aid. Moreover, he would not have great confidence in the number of his soldiers, and would not require reinforcements. Human nature was too much enamoured to allow men to render him service. \n\nWith the handful of soldiers on this side, and the army in apparently ill condition, the army was utterly destroyed." - ], - "training_time_seconds": 7167.931929588318, - "stage_training_flops": 1.3790348826629898e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.3790348826629898e+18, - "config_fingerprint": "a1e3ff2a651b87ea", - "git_commit_sha": "205cddabbb34257a8a78cf63a47ae281e3b51ac5", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/85989fb4", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r11.25-ctx4096", - "dataset_fingerprint": "63a5e6be81591d82", - "tokenizer_fingerprint": "7ad989ceea17794b", - "unique_train_tokens": 0 -} diff --git a/experiments/think-d12-r11.25-ctx4096/tokenizer/experiment_tokenizer.json b/experiments/think-d12-r11.25-ctx4096/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 5e164ca844ede497dc8f3a2287706ad19776d23e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "think-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1783526843 -} diff --git a/experiments/think-d12-r11.25-ctx4096/tokenizer/token_bytes.pt b/experiments/think-d12-r11.25-ctx4096/tokenizer/token_bytes.pt deleted file mode 100644 index 01d1ec4aab9e8a7d205c3b3ffbeb8da0e9a62db1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1 -size 132649 diff --git a/experiments/think-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl b/experiments/think-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl deleted file mode 100644 index a17bd392980021628053b95d6425fc556aad527a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1 -size 404071 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_000500.json b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_000500.json deleted file mode 100644 index f90f8873ef7208da04efc8c72725e8cc3a645313..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "think-d12-r11.25-ctx8192", - "val_bpb": 1.2806006574422366, - "model_config": { - "sequence_len": 8192, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-ctx8192", - "wandb_run_id": "e3483a4b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 8192, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 4, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints", - "experiment_id": "think-d12-r11.25-ctx8192", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json", - "tokenizer_fingerprint": "03c4f62e7a9d0c3b", - "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11.25-ctx8192", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx8192", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 8192, - "window_pattern": "L", - "device_batch_size": 4, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx8192", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx8192" - ] - }, - "config_fingerprint": "407a5074e0bf3730", - "artifact_path": "experiments/think-d12-r11.25-ctx8192" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx8192", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "407a5074e0bf3730" - }, - "device_batch_size": 4, - "max_seq_len": 8192, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.2806006574422366, - "smooth_train_loss": 3.5198863114259193, - "total_training_time": 1859.6971344947815, - "stage_training_flops": 410668272451584000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 410668272451584000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001000.json b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001000.json deleted file mode 100644 index ea0348fdef439360f2a2aaaac11e65d388f12e6a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "think-d12-r11.25-ctx8192", - "val_bpb": 1.1673074973592756, - "model_config": { - "sequence_len": 8192, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-ctx8192", - "wandb_run_id": "e3483a4b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 8192, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 4, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints", - "experiment_id": "think-d12-r11.25-ctx8192", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json", - "tokenizer_fingerprint": "03c4f62e7a9d0c3b", - "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11.25-ctx8192", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx8192", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 8192, - "window_pattern": "L", - "device_batch_size": 4, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx8192", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx8192" - ] - }, - "config_fingerprint": "407a5074e0bf3730", - "artifact_path": "experiments/think-d12-r11.25-ctx8192" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx8192", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "407a5074e0bf3730" - }, - "device_batch_size": 4, - "max_seq_len": 8192, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.1673074973592756, - "smooth_train_loss": 3.291130702382907, - "total_training_time": 3762.280524253845, - "stage_training_flops": 821336544903168000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 821336544903168000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001500.json b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001500.json deleted file mode 100644 index 17245f9d0ddd15b718a855106a0a3cee19388345..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "think-d12-r11.25-ctx8192", - "val_bpb": 1.108596107059193, - "model_config": { - "sequence_len": 8192, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-ctx8192", - "wandb_run_id": "e3483a4b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 8192, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 4, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints", - "experiment_id": "think-d12-r11.25-ctx8192", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json", - "tokenizer_fingerprint": "03c4f62e7a9d0c3b", - "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11.25-ctx8192", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx8192", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 8192, - "window_pattern": "L", - "device_batch_size": 4, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx8192", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx8192" - ] - }, - "config_fingerprint": "407a5074e0bf3730", - "artifact_path": "experiments/think-d12-r11.25-ctx8192" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx8192", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "407a5074e0bf3730" - }, - "device_batch_size": 4, - "max_seq_len": 8192, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.108596107059193, - "smooth_train_loss": 3.1279163553930784, - "total_training_time": 5663.0339615345, - "stage_training_flops": 1232004817354752000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1232004817354752000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002000.json b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002000.json deleted file mode 100644 index dc0ff369001a8ac804a19eebde85f9077dc18a0d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "think-d12-r11.25-ctx8192", - "val_bpb": 1.0589838913914653, - "model_config": { - "sequence_len": 8192, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-ctx8192", - "wandb_run_id": "e3483a4b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 8192, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 4, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints", - "experiment_id": "think-d12-r11.25-ctx8192", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json", - "tokenizer_fingerprint": "03c4f62e7a9d0c3b", - "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11.25-ctx8192", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx8192", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 8192, - "window_pattern": "L", - "device_batch_size": 4, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx8192", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx8192" - ] - }, - "config_fingerprint": "407a5074e0bf3730", - "artifact_path": "experiments/think-d12-r11.25-ctx8192" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx8192", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "407a5074e0bf3730" - }, - "device_batch_size": 4, - "max_seq_len": 8192, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.0589838913914653, - "smooth_train_loss": 3.116437961262795, - "total_training_time": 7566.1544008255005, - "stage_training_flops": 1642673089806336000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1642673089806336000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002362.json b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002362.json deleted file mode 100644 index b9c4c76cb8a461607a426151d0807d6328dcef59..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,140 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "think-d12-r11.25-ctx8192", - "val_bpb": 1.0395519592251896, - "model_config": { - "sequence_len": 8192, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r11.25-ctx8192", - "wandb_run_id": "e3483a4b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 8192, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 4, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 2000, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints", - "experiment_id": "think-d12-r11.25-ctx8192", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json", - "tokenizer_fingerprint": "03c4f62e7a9d0c3b", - "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r11.25-ctx8192", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx8192", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 8192, - "window_pattern": "L", - "device_batch_size": 4, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx8192", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx8192" - ] - }, - "config_fingerprint": "407a5074e0bf3730", - "artifact_path": "experiments/think-d12-r11.25-ctx8192" - }, - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx8192", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "407a5074e0bf3730" - }, - "device_batch_size": 4, - "max_seq_len": 8192, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38471586, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38471586 - }, - "loop_state": { - "min_val_bpb": 1.0395519592251896, - "smooth_train_loss": 3.0155535492313272, - "total_training_time": 9008.03685593605, - "stage_training_flops": 1939996919061282816, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1939996919061282816 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_000500.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_000500.pt deleted file mode 100644 index 69f1884a5327e9a0ed0e7a8f939c9443eb75b9e7..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:51922864ab7ccea4cd40458bb60898167d02696982c1e4c0de684995e5f26290 -size 792761690 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001000.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001000.pt deleted file mode 100644 index cba8efe605c7d58052dff025a32ac566cc343a2b..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c8c23459c7bef825a21a0236f2b9e2efba66c8b427c78a7a6cca852df957fd0e -size 792761690 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001500.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001500.pt deleted file mode 100644 index 3d9fee17f3c145eb8c8c6fd0413a26b73a45da30..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:5dd6cc545ba54787b342e190d75e857872d1964f10a4032ab0d49640012f84b7 -size 792761690 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002000.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002000.pt deleted file mode 100644 index 3be0f557534f7754337973fd2bdb3d2da5e582db..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d6e6c50126914bc4d20f9db6222891ee9ee61c634634b023a13e5f1e583e9403 -size 792761690 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002362.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002362.pt deleted file mode 100644 index 0aa26c24cc471452dc63a38e2a0a94011dd47672..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:43a07973a6d23613f28d1b43623e06bba393623fd88452e1c495bf26aa9ac4d6 -size 792761690 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 9f7ff257118d85ec5b935356ba451f8158fbdf2f..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f402cf80bf9d4b917ae2a99240b0ea34cd842f025927c148bccf384ce9b744b8 -size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 477768bd2b6531e7859664b868c3772d4c4fb848..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9ef8f97be5f4890871d5586afd42bc946b9058ae996deaa83c14c8f71610deb9 -size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 7d40d701879db1b27107cbb56c30a19f220e747d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:df9bf4a59d46c126bd65f5cdc6de8fb531c492fabf70f9db22a5ac27bd08dd2f -size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index bce34810607710181c5acee791e2beed95a1a712..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f95fc4fe7c0a841f3e443102d00495e4f4305ea955ad2b81ffd3ad8865cbd34e -size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002362_rank0.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 6e07dc627985d4520e9098d9170ff536446a3daa..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9e1464e581c921562375dd5957b76dad9fe37d04a17e9a021a7876026e624044 -size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx8192/config.json b/experiments/think-d12-r11.25-ctx8192/config.json deleted file mode 100644 index 71f8e9d529e0058c9b6ad1539cd12c5d9edb1780..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/config.json +++ /dev/null @@ -1,57 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25-ctx8192", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 8192, - "window_pattern": "L", - "device_batch_size": 4, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25-ctx8192", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25", - "ctx8192" - ] - }, - "config_fingerprint": "407a5074e0bf3730", - "artifact_path": "experiments/think-d12-r11.25-ctx8192" -} diff --git a/experiments/think-d12-r11.25-ctx8192/evals/core.json b/experiments/think-d12-r11.25-ctx8192/evals/core.json deleted file mode 100644 index feacd2f1e36f41ef00be8749074b5ac013f2579c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": 0.07251341817663091, - "core_results": { - "hellaswag_zeroshot": 0.2757418751716614, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.0832144096493721, - "arc_easy": 0.3194444477558136, - "arc_challenge": 0.21501706540584564, - "copa": 0.5099999904632568, - "commonsense_qa": 0.31285831332206726, - "piqa": 0.5331882238388062, - "openbook_qa": 0.24800001084804535, - "lambada_openai": 0.23423248529434204, - "hellaswag": 0.2802230417728424, - "winograd": 0.5714285969734192, - "winogrande": 0.4956590235233307, - "bigbench_dyck_languages": 0.10200000554323196, - "agi_eval_lsat_ar": 0.260869562625885, - "bigbench_cs_algorithms": 0.41969695687294006, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.024030273780226707, - "coqa": 0.0821746215224266, - "boolq": 0.5590214133262634, - "bigbench_language_identification": 0.2524999976158142 - }, - "centered_results": { - "hellaswag_zeroshot": 0.034322500228881836, - "jeopardy": 0.0009447330958209932, - "bigbench_qa_wikidata": 0.0832144096493721, - "arc_easy": 0.09259259700775146, - "arc_challenge": -0.04664391279220581, - "copa": 0.019999980926513672, - "commonsense_qa": 0.14107289165258405, - "piqa": 0.0663764476776123, - "openbook_qa": -0.002666652202606201, - "lambada_openai": 0.23423248529434204, - "hellaswag": 0.04029738903045654, - "winograd": 0.14285719394683838, - "winogrande": -0.008681952953338623, - "bigbench_dyck_languages": 0.10200000554323196, - "agi_eval_lsat_ar": 0.07608695328235625, - "bigbench_cs_algorithms": 0.41969695687294006, - "bigbench_operators": 0.07619047909975052, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.024030273780226707, - "coqa": 0.0821746215224266, - "boolq": -0.1604699649308857, - "bigbench_language_identification": 0.177667764153811 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/evals/samples.json b/experiments/think-d12-r11.25-ctx8192/evals/samples.json deleted file mode 100644 index f01d40c79c96e236727f04968d1a994c68c93574..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is not yet fully developed. The capital of the United States is not yet fully developed" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the gold, and the symbol of the silver. The gold is" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be the day of the week. \n\nThe day of the week is the same as" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as hot. \n\nThe hot is the same as hot. \n\nThe" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, 2. The moon, 3. The" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the color of the skin of the face, and the color of the skin." - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the" - } - ], - "unconditioned_samples": [ - "<|bos|>ALIENS. A concern called industrial, in which all miners of ability for useful labor were engaged in obtaining industrial materials for making snuffers. Although it would be a rude and untenable enterprise to make different classes of miners dispose of goods for profit, different miners differing between the quality of the material misspelled and its quantity, it always staggers the mind with the idea of the matter which it concerns.\u00b9 Four or five shopkeepers are seen at so many tradeshops in town near together in towns and villages. \n\nMoney is better paid to supply the needs of skilled men than it is in overcrowded", - "<|bos|>37374.31 617.471.11 379.75 \n\nSwinburne, Jes. 10, 335.\n\nStatistical Index. \n\nCoates, erance, 1333. \n\nPurple Debenture, 1460.\n\nGenetic Index. \n\nSwinfenning, 51858.29, 1319. \n\nFree Presses. \n\nRidgens-Pantrepous, 25. \n\nSprayl-Power, regular exercise, 1700. \n\nPreachers and Teachers of the Schools, 401\u20132. \n\n", - "<|bos|>URE FOOD AND THE BODY' \n\nIn such a commonwealth Siamese readers might find in Kumber's Essays, or Mabon's Vol. of Tobit and St. Jerome's Lives, a sound, a clear and satisfactory explanation of this phrase. May Lady Cassius inform the reverend Society from which this paragraph is borrowed that in this hour of peril men and women may \"paint to him,\" and bewail Rest and Treatment. In both of these circles there are three main meanings attached to the phrase. One is, with the excessive reference in hotel-keepers, donkeys, or hares; another, with alligators", - "<|bos|>HENRY MART 140. \n\nLeblay, Mr. De Martyn's invention of music, 4. IX.\n\nTRANSLATOR'S NOTE.-Send forth a translation of the Notes which were received by me, translated from the Musical \n\nCommission's Calendar.\n\nHis performance we cannot altogether estimate, but St. Columba gave us confirmation of his inventions, 30. XXVIIii, 18. Eh, What (Georgics, I, 177); 'Slightest book that ever was written' (Sonn., lies 28\u00bd), a work of high merit read with honour,\n\nQu\u00e6rese", - "<|bos|>Harvard School, IV Department of Education, 1843-1972. \n\n2 Henry State League, LL. concerning Courses in Medicine and the Arts, pp. 22 et seq.\n\nHistory of the Monroe Doctrine, by one who has visited Europe, compiled from European Authorities.\n\nNew York: N. Y. \n\nExaminer, Vol. XXXI, \n\nApril, 1917, p. 88.\n\nPamphlet on Scien tific Methods of Education (\"Outlines of the Maladies and Defects of the Methods of Industrial Society,\" by Dr. Jevons). \n\nNew York, September and October, 19", - "<|bos|>The Bird reflects upon. his. \n\nThe Clipper. \n\nVapour.\n\nUntil within a few weeks the admission of the truth to our beloved Bird was fatal to that race, little cared for neither in her recorded history nor since they married, and still less as regards her character. Her reign ended; and when she died only after a few months good for nothing the country felt herself well restored to health. She began now to see her way. Our dear bird became as dear to her as the Christian mother; she began to see her way clearer to her senses; and as her thoughts turned, and freedom fell back, she", - "<|bos|>Army of the Cumberland and Arkansas Army, and an Army of the Potomac under the command of Martin Robertson.\n\nHEADQUARTERS CAMP THIRDQUARTERS, THIRD BRIG 1ST BRIG 1ST BRIG 1ST BRIG 1ST BRIG \n\n6 8 8 \n\nMCLQUERISHER'S STATION, 9 P.M. \n\nMY ARMY, CAL. \n\nEnlarged with orders by the War Department.\n\nHeadquarters Camp War Department, Clope Ridge, Va., September 15, 1864. \n\n6 P.M The Confederates tend S. M. Camp are in the Confederate service hospital at N. C.", - "<|bos|>the eightieth year of his age.\n\nI had never been conferring, like the palette and the gallens at which I used to sit. never had thought it wrong to give the faintest hint of this prudery, which I trust is always requested of the upholder at his housekeeping, as shall appear by the order and directions accompanying it. So while I was listening to the wise old voice of the tender bride calling alone in her measure that nonsense of patriarchal impiety. It was all the more gratifying when I heard that the Governor of Shetton is now apparently labouring in the same breath, when he speaks" - ] -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/evals/val_bpb.json b/experiments/think-d12-r11.25-ctx8192/evals/val_bpb.json deleted file mode 100644 index 94e46b7e343d01e00db08cf58ca5b26a3c9e7b5d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/evals/val_bpb.json +++ /dev/null @@ -1,174 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val_per_position": [ - { - "start": 0, - "end": 256, - "bpb": 1.1345637945630735 - }, - { - "start": 256, - "end": 512, - "bpb": 1.0678589586934193 - }, - { - "start": 512, - "end": 768, - "bpb": 1.0513444442729074 - }, - { - "start": 768, - "end": 1024, - "bpb": 1.0420207610099703 - }, - { - "start": 1024, - "end": 1280, - "bpb": 1.0363256199642377 - }, - { - "start": 1280, - "end": 1536, - "bpb": 1.0293503892901525 - }, - { - "start": 1536, - "end": 1792, - "bpb": 1.0270513457476949 - }, - { - "start": 1792, - "end": 2048, - "bpb": 1.0255301775348173 - }, - { - "start": 2048, - "end": 2304, - "bpb": 1.0171632048394768 - }, - { - "start": 2304, - "end": 2560, - "bpb": 1.0150940036004366 - }, - { - "start": 2560, - "end": 2816, - "bpb": 1.0132597834490384 - }, - { - "start": 2816, - "end": 3072, - "bpb": 1.0154303287433495 - }, - { - "start": 3072, - "end": 3328, - "bpb": 1.0125085420227948 - }, - { - "start": 3328, - "end": 3584, - "bpb": 1.0117220799135607 - }, - { - "start": 3584, - "end": 3840, - "bpb": 1.0093052242865306 - }, - { - "start": 3840, - "end": 4096, - "bpb": 1.0067290549863526 - }, - { - "start": 4096, - "end": 4352, - "bpb": 1.007627760298138 - }, - { - "start": 4352, - "end": 4608, - "bpb": 1.005475773581573 - }, - { - "start": 4608, - "end": 4864, - "bpb": 1.0011348016292028 - }, - { - "start": 4864, - "end": 5120, - "bpb": 1.0025369118565095 - }, - { - "start": 5120, - "end": 5376, - "bpb": 0.9978363303279962 - }, - { - "start": 5376, - "end": 5632, - "bpb": 0.9936016054109623 - }, - { - "start": 5632, - "end": 5888, - "bpb": 0.9930789448128515 - }, - { - "start": 5888, - "end": 6144, - "bpb": 0.988120986726172 - }, - { - "start": 6144, - "end": 6400, - "bpb": 0.9878455863729801 - }, - { - "start": 6400, - "end": 6656, - "bpb": 0.988322568304946 - }, - { - "start": 6656, - "end": 6912, - "bpb": 0.9895607000315525 - }, - { - "start": 6912, - "end": 7168, - "bpb": 0.9924135354945043 - }, - { - "start": 7168, - "end": 7424, - "bpb": 0.9887797597227126 - }, - { - "start": 7424, - "end": 7680, - "bpb": 0.9863644464805957 - }, - { - "start": 7680, - "end": 7936, - "bpb": 0.9852729515784135 - }, - { - "start": 7936, - "end": 8192, - "bpb": 0.9838855089135268 - } - ], - "val": 1.0127322889400117 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/run.json b/experiments/think-d12-r11.25-ctx8192/run.json deleted file mode 100644 index c9dfef8e7cdc56887e670c7c0a925609d0ba770c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-r11.25-ctx8192", - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx8192", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "407a5074e0bf3730", - "wandb_run_id": "e3483a4b", - "created_at": 1783693257 -} diff --git a/experiments/think-d12-r11.25-ctx8192/summary.json b/experiments/think-d12-r11.25-ctx8192/summary.json deleted file mode 100644 index 3a2267f78ba4f11440074975edb5854408a02768..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/summary.json +++ /dev/null @@ -1,69 +0,0 @@ -{ - "experiment_id": "think-d12-r11.25-ctx8192", - "stage": "base", - "base_experiment_id": "think-d12-r11.25-ctx8192", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.0395519592251896, - "minimum_sampled_val_bpb": 1.0395519592251896, - "full_val_bpb": 1.0127322889400117, - "core_metric": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is not yet fully developed. The capital of the United States is not yet fully developed" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the symbol of the gold, and the symbol of the silver. The gold is" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be the day of the week. \n\nThe day of the week is the same as" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as hot. \n\nThe hot is the same as hot. \n\nThe" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, 2. The moon, 3. The" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the color of the skin of the face, and the color of the skin." - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the" - } - ], - "unconditioned_samples": [ - "<|bos|>ALIENS. A concern called industrial, in which all miners of ability for useful labor were engaged in obtaining industrial materials for making snuffers. Although it would be a rude and untenable enterprise to make different classes of miners dispose of goods for profit, different miners differing between the quality of the material misspelled and its quantity, it always staggers the mind with the idea of the matter which it concerns.\u00b9 Four or five shopkeepers are seen at so many tradeshops in town near together in towns and villages. \n\nMoney is better paid to supply the needs of skilled men than it is in overcrowded", - "<|bos|>37374.31 617.471.11 379.75 \n\nSwinburne, Jes. 10, 335.\n\nStatistical Index. \n\nCoates, erance, 1333. \n\nPurple Debenture, 1460.\n\nGenetic Index. \n\nSwinfenning, 51858.29, 1319. \n\nFree Presses. \n\nRidgens-Pantrepous, 25. \n\nSprayl-Power, regular exercise, 1700. \n\nPreachers and Teachers of the Schools, 401\u20132. \n\n", - "<|bos|>URE FOOD AND THE BODY' \n\nIn such a commonwealth Siamese readers might find in Kumber's Essays, or Mabon's Vol. of Tobit and St. Jerome's Lives, a sound, a clear and satisfactory explanation of this phrase. May Lady Cassius inform the reverend Society from which this paragraph is borrowed that in this hour of peril men and women may \"paint to him,\" and bewail Rest and Treatment. In both of these circles there are three main meanings attached to the phrase. One is, with the excessive reference in hotel-keepers, donkeys, or hares; another, with alligators", - "<|bos|>HENRY MART 140. \n\nLeblay, Mr. De Martyn's invention of music, 4. IX.\n\nTRANSLATOR'S NOTE.-Send forth a translation of the Notes which were received by me, translated from the Musical \n\nCommission's Calendar.\n\nHis performance we cannot altogether estimate, but St. Columba gave us confirmation of his inventions, 30. XXVIIii, 18. Eh, What (Georgics, I, 177); 'Slightest book that ever was written' (Sonn., lies 28\u00bd), a work of high merit read with honour,\n\nQu\u00e6rese", - "<|bos|>Harvard School, IV Department of Education, 1843-1972. \n\n2 Henry State League, LL. concerning Courses in Medicine and the Arts, pp. 22 et seq.\n\nHistory of the Monroe Doctrine, by one who has visited Europe, compiled from European Authorities.\n\nNew York: N. Y. \n\nExaminer, Vol. XXXI, \n\nApril, 1917, p. 88.\n\nPamphlet on Scien tific Methods of Education (\"Outlines of the Maladies and Defects of the Methods of Industrial Society,\" by Dr. Jevons). \n\nNew York, September and October, 19", - "<|bos|>The Bird reflects upon. his. \n\nThe Clipper. \n\nVapour.\n\nUntil within a few weeks the admission of the truth to our beloved Bird was fatal to that race, little cared for neither in her recorded history nor since they married, and still less as regards her character. Her reign ended; and when she died only after a few months good for nothing the country felt herself well restored to health. She began now to see her way. Our dear bird became as dear to her as the Christian mother; she began to see her way clearer to her senses; and as her thoughts turned, and freedom fell back, she", - "<|bos|>Army of the Cumberland and Arkansas Army, and an Army of the Potomac under the command of Martin Robertson.\n\nHEADQUARTERS CAMP THIRDQUARTERS, THIRD BRIG 1ST BRIG 1ST BRIG 1ST BRIG 1ST BRIG \n\n6 8 8 \n\nMCLQUERISHER'S STATION, 9 P.M. \n\nMY ARMY, CAL. \n\nEnlarged with orders by the War Department.\n\nHeadquarters Camp War Department, Clope Ridge, Va., September 15, 1864. \n\n6 P.M The Confederates tend S. M. Camp are in the Confederate service hospital at N. C.", - "<|bos|>the eightieth year of his age.\n\nI had never been conferring, like the palette and the gallens at which I used to sit. never had thought it wrong to give the faintest hint of this prudery, which I trust is always requested of the upholder at his housekeeping, as shall appear by the order and directions accompanying it. So while I was listening to the wise old voice of the tender bride calling alone in her measure that nonsense of patriarchal impiety. It was all the more gratifying when I heard that the Governor of Shetton is now apparently labouring in the same breath, when he speaks" - ], - "training_time_seconds": 9008.03685593605, - "stage_training_flops": 1.9399969190612828e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.9399969190612828e+18, - "config_fingerprint": "407a5074e0bf3730", - "git_commit_sha": "083cd7f99484b5e894a23e4a093de6d339c412ea", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/e3483a4b", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r11.25-ctx8192", - "dataset_fingerprint": "63a5e6be81591d82", - "tokenizer_fingerprint": "03c4f62e7a9d0c3b", - "unique_train_tokens": 0 -} diff --git a/experiments/think-d12-r11.25-ctx8192/tokenizer/experiment_tokenizer.json b/experiments/think-d12-r11.25-ctx8192/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 8a28b84314d5995d3b64f34451ce3a694d901f80..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "think-d12-r11.25-ctx8192", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1783708207 -} diff --git a/experiments/think-d12-r11.25-ctx8192/tokenizer/token_bytes.pt b/experiments/think-d12-r11.25-ctx8192/tokenizer/token_bytes.pt deleted file mode 100644 index 01d1ec4aab9e8a7d205c3b3ffbeb8da0e9a62db1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1 -size 132649 diff --git a/experiments/think-d12-r11.25-ctx8192/tokenizer/tokenizer.pkl b/experiments/think-d12-r11.25-ctx8192/tokenizer/tokenizer.pkl deleted file mode 100644 index a17bd392980021628053b95d6425fc556aad527a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25-ctx8192/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1 -size 404071 diff --git a/experiments/think-d12-r11.25/base_checkpoints/meta_000500.json b/experiments/think-d12-r11.25/base_checkpoints/meta_000500.json deleted file mode 100644 index 4346b0a0961c0461b449f5bf3e53e0e367f9f980..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 500, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": null - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.5893779623775406, - "total_training_time": 1290.4532148838043 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/base_checkpoints/meta_001000.json b/experiments/think-d12-r11.25/base_checkpoints/meta_001000.json deleted file mode 100644 index d44aff60900e63f44ee57a352b66725c582f2c36..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 1000, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": null - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.4419165825253133, - "total_training_time": 2609.577807664871 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/base_checkpoints/meta_001500.json b/experiments/think-d12-r11.25/base_checkpoints/meta_001500.json deleted file mode 100644 index 1991d9750741add910d01954f8482d4e2247e360..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 1500, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": null - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.2465426140603015, - "total_training_time": 3929.598204135895 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/base_checkpoints/meta_002000.json b/experiments/think-d12-r11.25/base_checkpoints/meta_002000.json deleted file mode 100644 index 488a6f07b4fea050f2e89f34893aad59564826a0..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 2000, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": null - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.2442641345309373, - "total_training_time": 5249.494728565216 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/base_checkpoints/meta_002362.json b/experiments/think-d12-r11.25/base_checkpoints/meta_002362.json deleted file mode 100644 index e175df81e356e808f571537ad3eb29ed517c3335..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 2362, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": null - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.074421420856799, - "total_training_time": 6205.646646976471 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/base_checkpoints/model_000500.pt b/experiments/think-d12-r11.25/base_checkpoints/model_000500.pt deleted file mode 100644 index adb2ea235f50e379dec7aedafe47be61861bedab..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:21fbdff366e5db3fa2f254b4b98c763cc70ba95722242fa032d8d21b956b694f -size 792761399 diff --git a/experiments/think-d12-r11.25/base_checkpoints/model_001000.pt b/experiments/think-d12-r11.25/base_checkpoints/model_001000.pt deleted file mode 100644 index 6c50be693db2de57ed0fc9edeb4808310b2631ab..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6f0a862044e51b8cac250cf1e4f26af7f8347fd887b8c30af08addb879b6494d -size 792761399 diff --git a/experiments/think-d12-r11.25/base_checkpoints/model_001500.pt b/experiments/think-d12-r11.25/base_checkpoints/model_001500.pt deleted file mode 100644 index 6c11f7b0d9217759f1212acf26a448949fffd5d8..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:326bddc26c100de33310f9164dc873af6099a9b7760527a8707c256988a8ec7a -size 792761399 diff --git a/experiments/think-d12-r11.25/base_checkpoints/model_002000.pt b/experiments/think-d12-r11.25/base_checkpoints/model_002000.pt deleted file mode 100644 index 4ec265330fa790437406e93a5448ad62eb39fe99..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a5d101975722cffc43eb4f35aa967a5afd397b159da3521e5a1acbd3819d466e -size 792761399 diff --git a/experiments/think-d12-r11.25/base_checkpoints/model_002362.pt b/experiments/think-d12-r11.25/base_checkpoints/model_002362.pt deleted file mode 100644 index 6cf06b066423745b22aa44519e4b864d5bdc27d6..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:04331c52ed8fa7259f350e4ec72d0dd6602451cfd75a5773a4c17ac5c141ea7e -size 792761399 diff --git a/experiments/think-d12-r11.25/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r11.25/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 79c1822fb2dc78fb93423cdc8813dfc59fea2a94..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d39a131089a476a202ae932f21d9398b225d0b8e4147a58fbdff797914d34976 -size 1246165237 diff --git a/experiments/think-d12-r11.25/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r11.25/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index f055caef01d26659f49ac7ae7f17083f45d82d4a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e4971566eb3d4aa5d5fe29b3a3b77f56581b61713e5c1d162debb8a409c02118 -size 1246165237 diff --git a/experiments/think-d12-r11.25/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r11.25/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 275f823331fcb1cd614bc3d8e15f97265d98d85a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:05ce87d44bfd6e9b4d4c1e643a2a5aa4b169119913ccbdf3625d7cfa7313b342 -size 1246165237 diff --git a/experiments/think-d12-r11.25/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r11.25/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index b3b5e1bd94301a192b14e234e17f4f28786b5538..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1b58f68fd887360ce99e8456aecf706b1bcf5c8210194721389698ec658237b1 -size 1246165237 diff --git a/experiments/think-d12-r11.25/base_checkpoints/optim_002362_rank0.pt b/experiments/think-d12-r11.25/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 47439d09d2d9170f2458f5ad529c461a8e8c7f80..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:09aa5632a4b33981a1b2fbd97254d0050ecb7916d0c242f1d195c0726be314d7 -size 1246165237 diff --git a/experiments/think-d12-r11.25/config.json b/experiments/think-d12-r11.25/config.json deleted file mode 100644 index 3d25cc9589b19a727199d6a7f46b93beb73eff45..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/config.json +++ /dev/null @@ -1,55 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r11.25", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r11.25", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "3b5a68714770b6af", - "artifact_path": "experiments/think-d12-r11.25" -} diff --git a/experiments/think-d12-r11.25/evals/samples.json b/experiments/think-d12-r11.25/evals/samples.json deleted file mode 100644 index 577869f25a98c31214b820db1b56d340add64022..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is 10,000,000 francs, and the capital of the United" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the gold of the \n\nUnited States. It is the gold of the United States" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nThe day is Sunday, and the day is Sunday. \n\nThe" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the opposite of cold. \n\nThe opposite of cold is the opposite of cold." - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, and the stars. \n\n2." - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the color of the sky. \n\nThe color of the sky is a color of" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times the number of times the number of times the number of times" - } - ], - "unconditioned_samples": [ - "<|bos|>Monthillahivan-this is worthy, they say, of the glory of God's presence, our Father being now to come again! \n\nDead. The priest who said Lord! so between thy guilty hands exulting and curse shoot; do thy honors sweetly, O Father! touch me not with panic, nor miss me with holy surprise, nor sigh (say) tost away my days, nor desolate them; thou wert sent by heaven, and seen with so much care, that Thou near thy sins didst repair them; how art Thou now to come thus early, and bear unto me so divinely? not", - "<|bos|>370 \n\nPaxton's Introduction to Education, or the true philosophy of the schools criticised, is differentiated. In Rossetti's plan, as already noted, it formulated in these brief articles, there should be but two different versions of the 73\n\n-flat elements; in Rossetti's, we commonly use the broad current of their general aim. That which has been called a positive and material-is aptly called a negative-was given with corresponding emphasis at three different times, at different epochs, but the THREE are frequently gods regular in origin, and preserved from the violence of adjustment and shift, rather than dormant and deliberate meaning of", - "<|bos|>ROrepoys and Willieot's book on the Siamese Law. \n\nA Letter from Frank Perrivsky to a Prince of Wales. By Kinsman Keith. \n\nTHE \n\nMAY, 1923. \n\nIn Paper \n\nWith 25 Illustrations. Parts. $2 $9 $14 6 $2.50 \n\nIN POLITICAL IDOLSES. \n\nAmerican Law. By George W. Resting, LL.D., Professor of Political Economy in Princeton University. \n\n$18 $2.50 \n\nSamson Minot, hotel-keeper. $3 $7\n\nTHE SCALE OF SIZE.", - "<|bos|>HENRY MARTYN SAXON, THE HINDUS CHRISTURIENT. \n\nEDWARD IRVING, of the College of the Anatomy School of \n\nDurham, Surrey.\n\nEDWARD PERCY BAKER, OF ALICE COLLEGE, whose personal appearance is now in print, was born at Stanneley, Surrey, July 18, 1794. He has been student in the Company's Military College at Woolwich for more than one year, for his learning and industry in his profession and studies. He has written a book entitled History, Economics and Political Science, which lies nearly at our very door, entitled History, Political Science and Political \n\nScience. It is", - "<|bos|> HOUSE OF THE ANGELS. \n\nFrom the City of St. Ann. \n\n2 vols. 3s. My Last in a Garden. I reserve for the fifth edition, in manuscript, a full account of ancient His tory. In \n\n1 vol. 5, a short history of Nero and Herod. from 6 to \n\n10 Years, both kept in this Library.\n\n4 \n\nLibrary of the Inducci EN QUANDRON. \n\nFRONTIER, SAMUEL, Dean of Carlisle, President of the\n\nAmerican Board of Works, 3 vols. 2 vols. 3s. 6d. \n\nLibrary", - "<|bos|>The Bird reflects the world, and swims the eagle's web.-Van Isle.\n\nSHOP'S \n\nREACH \n\nNEGARYELY GENIAL SHOP FINANCIERS RAIDER \n\nREAR HEADS,.} \n\nREPUBLICS, WHILE SHE UNDER FULLER'S PORT, \n\nREP Philosophers, that they be not \n\nPharaoh's patterns good for nothing, and valiant men for that which is nothing; \n\nCLY VAUS\u00d2 EXOVENT \u03b4\u1f72 \u03bf\u03cd\u03c3\u03b1\u03b9 \u0391\u03b4\u03af\u03b4\u03b5\u03b9ANTA\u03c1, \u1f15\u03b4\u03b1\u03c1\u03c9\u03bd \u03c3\u03bf\u03c5\u03b4\u03bf\u1f7a\u03c2 The Queen eschews", - "<|bos|>'Eau du Monarch beau ou H\u00f4tel de Voodjiches.\" (The same French Inspecteur :) \"Son jours\n\nA \u00e9t\u00e9 autant \u00e0 tomboi les m\u00eames Premi\u00e8res avec une princesse qui sont d'ombres le miraculeur bien de France le souffrir. Les c\u00f4tes de ceux qui le sont j\u00e9sibres.\" (The French Directors.)\n\neffected these river improvements in effeminate cases, in the The Executive has eschewed the corrupdemell\u00e8 hrs. Nieuw Ga", - "<|bos|>the eight (250) series.\n\nThe pressure is already great.\n\nashions like the palette de la \n\nLast chapter, however, it may be noted, and it is a most common practice in such a period for small and countless series to transform the palette de la up half at the sauce, and it is a matter of considerable importance to seeing the clothes-holders uniformly the marks that indicate the supply of the palette de la into that respectable width whence their sulky tints are generally returned. \n\nThe colour represents the price of the garment at the time apparently labouring in the work. Delivery is possible" - ] -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/evals/val_bpb.json b/experiments/think-d12-r11.25/evals/val_bpb.json deleted file mode 100644 index 34f6714207d8b0ee7715eeda9acd0ca344802cf2..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/evals/val_bpb.json +++ /dev/null @@ -1,54 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val_per_position": [ - { - "start": 0, - "end": 256, - "bpb": 1.1255410237731462 - }, - { - "start": 256, - "end": 512, - "bpb": 1.0636677720059344 - }, - { - "start": 512, - "end": 768, - "bpb": 1.0506832286096848 - }, - { - "start": 768, - "end": 1024, - "bpb": 1.0456638664028364 - }, - { - "start": 1024, - "end": 1280, - "bpb": 1.038324560694976 - }, - { - "start": 1280, - "end": 1536, - "bpb": 1.0330914959872517 - }, - { - "start": 1536, - "end": 1792, - "bpb": 1.0308104068466784 - }, - { - "start": 1792, - "end": 2048, - "bpb": 1.027579795687179 - } - ], - "val": 1.0519195678472355 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean-1930s.json b/experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean-1930s.json deleted file mode 100644 index aa836c8edd73b612a1cc9c69196ef0398a23fd24..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean-1930s.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.0781689302026238 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json b/experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json deleted file mode 100644 index 9fc110ec140db1d49acaca9fd71b69b47fcb85dd..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.081385374786877 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/run.json b/experiments/think-d12-r11.25/run.json deleted file mode 100644 index 1004e46b9b3dd13da3a3c8aab534df801a665cc8..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/run.json +++ /dev/null @@ -1,6 +0,0 @@ -{ - "experiment_id": "think-d12-r11.25", - "stage": "base", - "wandb_run_id": null, - "migration_note": "Migrated from the pre-lineage repository layout." -} diff --git a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/meta_001065.json b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/meta_001065.json deleted file mode 100644 index 2f52697c9b62b9b51836bb31924aadd0d871459c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/meta_001065.json +++ /dev/null @@ -1,38 +0,0 @@ -{ - "step": 1065, - "val_bpb": 0.39271438585109175, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "model_tag": "d12", - "model_step": null, - "load_optimizer": 1, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.0, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": -1, - "eval_tokens": 20971520, - "chatcore_every": -1, - "chatcore_max_cat": -1, - "chatcore_max_sample": 24, - "mmlu_epochs": 3, - "gsm8k_epochs": 4 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/model_001065.pt b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/model_001065.pt deleted file mode 100644 index 637ffa7aa73462ac49b2503b5b34b31bfcd7b3ce..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/model_001065.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f1ef8886ea2cbd820baa9bc98368f99673efb1f5fdbd082196d573b291111bda -size 792761399 diff --git a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/optim_001065_rank0.pt b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/optim_001065_rank0.pt deleted file mode 100644 index 451dddf0789f3113d3b31c3792c2328b1f92da2c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/optim_001065_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:db8c9df6e288618ed6362102e29edc3731267b4ff382154a55709b4d5a14f1d4 -size 1246165237 diff --git a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/config.json b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/config.json deleted file mode 100644 index cd6f5cd8b6a44a9a1d15b1422d203728a7bdfd83..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/config.json +++ /dev/null @@ -1,36 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_id": "smoltalk-mmlu3-gsm8k4-v1", - "parent": { - "base_experiment_id": "think-d12-r11.25", - "checkpoint_step": 2362 - }, - "data": { - "recipe": "nanochat-default", - "mmlu_epochs": 3, - "gsm8k_epochs": 4 - }, - "training": { - "num_iterations": -1, - "device_batch_size": 8, - "eval_every": -1, - "chatcore_every": -1, - "save_every": 200 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": false, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "smoltalk-mmlu3-gsm8k4-v1", - "group": "think-d12-r11.25", - "tags": ["sft", "smoltalk", "mmlu3", "gsm8k4"] - }, - "historical": { - "lineage_inferred_from_checkpoint_metadata": true, - "wandb_was_disabled": true - } -} diff --git a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/run.json b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/run.json deleted file mode 100644 index 4f172aadf1e6e2711e0b17ef0257df8bfdc9486e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/run.json +++ /dev/null @@ -1,6 +0,0 @@ -{ - "experiment_id": "smoltalk-mmlu3-gsm8k4-v1", - "stage": "sft", - "wandb_run_id": null, - "migration_note": "Migrated from the pre-lineage repository layout." -} diff --git a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/summary.json b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/summary.json deleted file mode 100644 index 50e2e8ee27321ad7b1c85abc1f6f87bbf3704314..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/summary.json +++ /dev/null @@ -1,14 +0,0 @@ -{ - "stage": "sft", - "experiment_id": "smoltalk-mmlu3-gsm8k4-v1", - "base_experiment_id": "think-d12-r11.25", - "parent_experiment_id": "think-d12-r11.25", - "parent_checkpoint_step": 2362, - "checkpoint_step": 1065, - "training_tokens": 558366720, - "flops_per_token": 887097900.0, - "stage_training_flops": 4.95325944741888e+17, - "inherited_parent_flops": 1.0985538793242624e+18, - "cumulative_pipeline_training_flops": 1.5938798240661504e+18, - "lineage_inferred_from_checkpoint_metadata": true -} diff --git a/experiments/think-d12-r11.25/summary.json b/experiments/think-d12-r11.25/summary.json deleted file mode 100644 index 7da472fc36723c48d2b604c030abf02af4ee789c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/summary.json +++ /dev/null @@ -1,69 +0,0 @@ -{ - "experiment_id": "think-d12-r11.25", - "stage": "base", - "base_experiment_id": "think-d12-r11.25", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": null, - "minimum_sampled_val_bpb": Infinity, - "full_val_bpb": 1.0519195678472355, - "core_metric": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is 10,000,000 francs, and the capital of the United" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the gold of the \n\nUnited States. It is the gold of the United States" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nThe day is Sunday, and the day is Sunday. \n\nThe" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the opposite of cold. \n\nThe opposite of cold is the opposite of cold." - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, and the stars. \n\n2." - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the color of the sky. \n\nThe color of the sky is a color of" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times the number of times the number of times the number of times" - } - ], - "unconditioned_samples": [ - "<|bos|>Monthillahivan-this is worthy, they say, of the glory of God's presence, our Father being now to come again! \n\nDead. The priest who said Lord! so between thy guilty hands exulting and curse shoot; do thy honors sweetly, O Father! touch me not with panic, nor miss me with holy surprise, nor sigh (say) tost away my days, nor desolate them; thou wert sent by heaven, and seen with so much care, that Thou near thy sins didst repair them; how art Thou now to come thus early, and bear unto me so divinely? not", - "<|bos|>370 \n\nPaxton's Introduction to Education, or the true philosophy of the schools criticised, is differentiated. In Rossetti's plan, as already noted, it formulated in these brief articles, there should be but two different versions of the 73\n\n-flat elements; in Rossetti's, we commonly use the broad current of their general aim. That which has been called a positive and material-is aptly called a negative-was given with corresponding emphasis at three different times, at different epochs, but the THREE are frequently gods regular in origin, and preserved from the violence of adjustment and shift, rather than dormant and deliberate meaning of", - "<|bos|>ROrepoys and Willieot's book on the Siamese Law. \n\nA Letter from Frank Perrivsky to a Prince of Wales. By Kinsman Keith. \n\nTHE \n\nMAY, 1923. \n\nIn Paper \n\nWith 25 Illustrations. Parts. $2 $9 $14 6 $2.50 \n\nIN POLITICAL IDOLSES. \n\nAmerican Law. By George W. Resting, LL.D., Professor of Political Economy in Princeton University. \n\n$18 $2.50 \n\nSamson Minot, hotel-keeper. $3 $7\n\nTHE SCALE OF SIZE.", - "<|bos|>HENRY MARTYN SAXON, THE HINDUS CHRISTURIENT. \n\nEDWARD IRVING, of the College of the Anatomy School of \n\nDurham, Surrey.\n\nEDWARD PERCY BAKER, OF ALICE COLLEGE, whose personal appearance is now in print, was born at Stanneley, Surrey, July 18, 1794. He has been student in the Company's Military College at Woolwich for more than one year, for his learning and industry in his profession and studies. He has written a book entitled History, Economics and Political Science, which lies nearly at our very door, entitled History, Political Science and Political \n\nScience. It is", - "<|bos|> HOUSE OF THE ANGELS. \n\nFrom the City of St. Ann. \n\n2 vols. 3s. My Last in a Garden. I reserve for the fifth edition, in manuscript, a full account of ancient His tory. In \n\n1 vol. 5, a short history of Nero and Herod. from 6 to \n\n10 Years, both kept in this Library.\n\n4 \n\nLibrary of the Inducci EN QUANDRON. \n\nFRONTIER, SAMUEL, Dean of Carlisle, President of the\n\nAmerican Board of Works, 3 vols. 2 vols. 3s. 6d. \n\nLibrary", - "<|bos|>The Bird reflects the world, and swims the eagle's web.-Van Isle.\n\nSHOP'S \n\nREACH \n\nNEGARYELY GENIAL SHOP FINANCIERS RAIDER \n\nREAR HEADS,.} \n\nREPUBLICS, WHILE SHE UNDER FULLER'S PORT, \n\nREP Philosophers, that they be not \n\nPharaoh's patterns good for nothing, and valiant men for that which is nothing; \n\nCLY VAUS\u00d2 EXOVENT \u03b4\u1f72 \u03bf\u03cd\u03c3\u03b1\u03b9 \u0391\u03b4\u03af\u03b4\u03b5\u03b9ANTA\u03c1, \u1f15\u03b4\u03b1\u03c1\u03c9\u03bd \u03c3\u03bf\u03c5\u03b4\u03bf\u1f7a\u03c2 The Queen eschews", - "<|bos|>'Eau du Monarch beau ou H\u00f4tel de Voodjiches.\" (The same French Inspecteur :) \"Son jours\n\nA \u00e9t\u00e9 autant \u00e0 tomboi les m\u00eames Premi\u00e8res avec une princesse qui sont d'ombres le miraculeur bien de France le souffrir. Les c\u00f4tes de ceux qui le sont j\u00e9sibres.\" (The French Directors.)\n\neffected these river improvements in effeminate cases, in the The Executive has eschewed the corrupdemell\u00e8 hrs. Nieuw Ga", - "<|bos|>the eight (250) series.\n\nThe pressure is already great.\n\nashions like the palette de la \n\nLast chapter, however, it may be noted, and it is a most common practice in such a period for small and countless series to transform the palette de la up half at the sauce, and it is a matter of considerable importance to seeing the clothes-holders uniformly the marks that indicate the supply of the palette de la into that respectable width whence their sulky tints are generally returned. \n\nThe colour represents the price of the garment at the time apparently labouring in the work. Delivery is possible" - ], - "training_time_seconds": 6205.646646976471, - "stage_training_flops": 1.0985538793242624e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.0985538793242624e+18, - "config_fingerprint": "3b5a68714770b6af", - "git_commit_sha": "205cddabbb34257a8a78cf63a47ae281e3b51ac5", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/None", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r11.25", - "dataset_fingerprint": "63a5e6be81591d82", - "tokenizer_fingerprint": "6e592b9b323f98bf", - "unique_train_tokens": 0 -} diff --git a/experiments/think-d12-r11.25/tokenizer/think_dataset_tokenizer.json b/experiments/think-d12-r11.25/tokenizer/think_dataset_tokenizer.json deleted file mode 100644 index fea3141f8940ad53370916cd7a3dc8743edbf3ad..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/tokenizer/think_dataset_tokenizer.json +++ /dev/null @@ -1,20 +0,0 @@ -{ - "dataset_repo": "jbduran/think-dataset", - "manifest_source_dataset": "institutional/institutional-books-1.0", - "num_train_shards": 24, - "expected_val_shard": "shard_00472.parquet", - "filters": { - "language": "eng", - "min_english_proportion": 0.9, - "year_max_exclusive": 1930, - "reject_invalid_date_types": true, - "ocr_min_inclusive": 90.0, - "ocr_disagreement_max_inclusive": 10.0, - "min_tokenizability": 95.0, - "min_tokens": 500, - "min_chars": 2000, - "min_pages": 3, - "min_sentences": 20, - "undated_rows_rejected": true - } -} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/tokenizer/token_bytes.pt b/experiments/think-d12-r11.25/tokenizer/token_bytes.pt deleted file mode 100644 index 01d1ec4aab9e8a7d205c3b3ffbeb8da0e9a62db1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1 -size 132649 diff --git a/experiments/think-d12-r11.25/tokenizer/tokenizer.pkl b/experiments/think-d12-r11.25/tokenizer/tokenizer.pkl deleted file mode 100644 index a17bd392980021628053b95d6425fc556aad527a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r11.25/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1 -size 404071 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_000500.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_000500.json deleted file mode 100644 index 5c847924bceeb560949ff7f6cebd8cd50165b349..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,137 +0,0 @@ -{ - "step": 500, - "experiment_id": "think-d12-r20-2epoch", - "val_bpb": 1.3372388496121728, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-2epoch", - "wandb_run_id": "520f145b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,2epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 4995, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", - "experiment_id": "think-d12-r20-2epoch", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", - "tokenizer_fingerprint": "90b338f6bf263273", - "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-2epoch", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-2epoch", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 22, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "epochs": 2, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core-metric-max-per-task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-2epoch", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "2epoch" - ] - }, - "config_fingerprint": "23f3b18820c5ee2b", - "artifact_path": "experiments/think-d12-r20-2epoch" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-2epoch", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "23f3b18820c5ee2b" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3372388496121728, - "smooth_train_loss": 3.667125674624101, - "total_training_time": 1320.8929188251495, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_001000.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_001000.json deleted file mode 100644 index d542a7ac0105130d8ccb3f2df7e1e8419a91022a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,137 +0,0 @@ -{ - "step": 1000, - "experiment_id": "think-d12-r20-2epoch", - "val_bpb": 1.2543626382469548, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-2epoch", - "wandb_run_id": "520f145b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,2epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 4995, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", - "experiment_id": "think-d12-r20-2epoch", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", - "tokenizer_fingerprint": "90b338f6bf263273", - "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-2epoch", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-2epoch", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 22, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "epochs": 2, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core-metric-max-per-task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-2epoch", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "2epoch" - ] - }, - "config_fingerprint": "23f3b18820c5ee2b", - "artifact_path": "experiments/think-d12-r20-2epoch" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-2epoch", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "23f3b18820c5ee2b" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2543626382469548, - "smooth_train_loss": 3.4484858640243488, - "total_training_time": 2668.93452835083, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_001500.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_001500.json deleted file mode 100644 index c6d5fa29bfcfba7daaa51791e245beb3a8f2f57e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,137 +0,0 @@ -{ - "step": 1500, - "experiment_id": "think-d12-r20-2epoch", - "val_bpb": 1.2340784059707743, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-2epoch", - "wandb_run_id": "520f145b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,2epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 4995, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", - "experiment_id": "think-d12-r20-2epoch", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", - "tokenizer_fingerprint": "90b338f6bf263273", - "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-2epoch", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-2epoch", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 22, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "epochs": 2, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core-metric-max-per-task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-2epoch", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "2epoch" - ] - }, - "config_fingerprint": "23f3b18820c5ee2b", - "artifact_path": "experiments/think-d12-r20-2epoch" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-2epoch", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "23f3b18820c5ee2b" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.2340784059707743, - "smooth_train_loss": 3.6208812604507203, - "total_training_time": 4014.0752940177917, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_002000.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_002000.json deleted file mode 100644 index e5ce8769b531e47a67ba81a47fb6222c5dcc9c96..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,137 +0,0 @@ -{ - "step": 2000, - "experiment_id": "think-d12-r20-2epoch", - "val_bpb": 1.2034096822606946, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-2epoch", - "wandb_run_id": "520f145b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,2epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 4995, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", - "experiment_id": "think-d12-r20-2epoch", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", - "tokenizer_fingerprint": "90b338f6bf263273", - "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-2epoch", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-2epoch", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 22, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "epochs": 2, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core-metric-max-per-task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-2epoch", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "2epoch" - ] - }, - "config_fingerprint": "23f3b18820c5ee2b", - "artifact_path": "experiments/think-d12-r20-2epoch" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-2epoch", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "23f3b18820c5ee2b" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.2034096822606946, - "smooth_train_loss": 3.433083085366257, - "total_training_time": 5361.500252485275, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_002500.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_002500.json deleted file mode 100644 index ffc9a73ca481391c1d9f57d019eac3a72d81196b..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_002500.json +++ /dev/null @@ -1,137 +0,0 @@ -{ - "step": 2500, - "experiment_id": "think-d12-r20-2epoch", - "val_bpb": 1.1723409617937781, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-2epoch", - "wandb_run_id": "520f145b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,2epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 4995, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", - "experiment_id": "think-d12-r20-2epoch", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", - "tokenizer_fingerprint": "90b338f6bf263273", - "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-2epoch", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-2epoch", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 22, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "epochs": 2, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core-metric-max-per-task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-2epoch", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "2epoch" - ] - }, - "config_fingerprint": "23f3b18820c5ee2b", - "artifact_path": "experiments/think-d12-r20-2epoch" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-2epoch", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "23f3b18820c5ee2b" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 0, - "pos": 1487218, - "epoch": 2, - "pq_idx": 0, - "rg_idx": 1487218 - }, - "loop_state": { - "min_val_bpb": 1.1723409617937781, - "smooth_train_loss": 3.450269059013645, - "total_training_time": 6707.027107000351, - "stage_training_flops": 1162736943759360000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1162736943759360000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_003000.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_003000.json deleted file mode 100644 index 51cf4e29f6ad883df2dbff79b353d81180f720c8..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_003000.json +++ /dev/null @@ -1,137 +0,0 @@ -{ - "step": 3000, - "experiment_id": "think-d12-r20-2epoch", - "val_bpb": 1.1416312127753048, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-2epoch", - "wandb_run_id": "520f145b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,2epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 4995, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", - "experiment_id": "think-d12-r20-2epoch", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", - "tokenizer_fingerprint": "90b338f6bf263273", - "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-2epoch", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-2epoch", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 22, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "epochs": 2, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core-metric-max-per-task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-2epoch", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "2epoch" - ] - }, - "config_fingerprint": "23f3b18820c5ee2b", - "artifact_path": "experiments/think-d12-r20-2epoch" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-2epoch", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "23f3b18820c5ee2b" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 63639218, - "epoch": 2, - "pq_idx": 2, - "rg_idx": 63639218 - }, - "loop_state": { - "min_val_bpb": 1.1416312127753048, - "smooth_train_loss": 3.17850348486724, - "total_training_time": 8053.040769100189, - "stage_training_flops": 1395284332511232000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1395284332511232000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_003500.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_003500.json deleted file mode 100644 index 652173eeddc45eca75bc9f733eb1049ae68f7dc6..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_003500.json +++ /dev/null @@ -1,137 +0,0 @@ -{ - "step": 3500, - "experiment_id": "think-d12-r20-2epoch", - "val_bpb": 1.118717862479829, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-2epoch", - "wandb_run_id": "520f145b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,2epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 4995, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", - "experiment_id": "think-d12-r20-2epoch", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", - "tokenizer_fingerprint": "90b338f6bf263273", - "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-2epoch", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-2epoch", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 22, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "epochs": 2, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core-metric-max-per-task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-2epoch", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "2epoch" - ] - }, - "config_fingerprint": "23f3b18820c5ee2b", - "artifact_path": "experiments/think-d12-r20-2epoch" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-2epoch", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "23f3b18820c5ee2b" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 25791218, - "epoch": 2, - "pq_idx": 5, - "rg_idx": 25791218 - }, - "loop_state": { - "min_val_bpb": 1.118717862479829, - "smooth_train_loss": 3.0035985624321593, - "total_training_time": 9400.758259773254, - "stage_training_flops": 1627831721263104000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1627831721263104000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004000.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004000.json deleted file mode 100644 index e2d8e42f0a42f7323e12c546a8973aaff14ba5f2..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004000.json +++ /dev/null @@ -1,137 +0,0 @@ -{ - "step": 4000, - "experiment_id": "think-d12-r20-2epoch", - "val_bpb": 1.0983701045807681, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-2epoch", - "wandb_run_id": "520f145b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,2epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 4995, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", - "experiment_id": "think-d12-r20-2epoch", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", - "tokenizer_fingerprint": "90b338f6bf263273", - "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-2epoch", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-2epoch", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 22, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "epochs": 2, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core-metric-max-per-task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-2epoch", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "2epoch" - ] - }, - "config_fingerprint": "23f3b18820c5ee2b", - "artifact_path": "experiments/think-d12-r20-2epoch" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-2epoch", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "23f3b18820c5ee2b" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 87943218, - "epoch": 2, - "pq_idx": 7, - "rg_idx": 87943218 - }, - "loop_state": { - "min_val_bpb": 1.0983701045807681, - "smooth_train_loss": 3.169070686059073, - "total_training_time": 10746.176971197128, - "stage_training_flops": 1860379110014976000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1860379110014976000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004500.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004500.json deleted file mode 100644 index 66ce6157e686363fa041bea5c0722b2fddb60580..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004500.json +++ /dev/null @@ -1,137 +0,0 @@ -{ - "step": 4500, - "experiment_id": "think-d12-r20-2epoch", - "val_bpb": 1.0766404457414214, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-2epoch", - "wandb_run_id": "520f145b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,2epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 4995, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", - "experiment_id": "think-d12-r20-2epoch", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", - "tokenizer_fingerprint": "90b338f6bf263273", - "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-2epoch", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-2epoch", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 22, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "epochs": 2, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core-metric-max-per-task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-2epoch", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "2epoch" - ] - }, - "config_fingerprint": "23f3b18820c5ee2b", - "artifact_path": "experiments/think-d12-r20-2epoch" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-2epoch", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "23f3b18820c5ee2b" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 50095218, - "epoch": 2, - "pq_idx": 10, - "rg_idx": 50095218 - }, - "loop_state": { - "min_val_bpb": 1.0766404457414214, - "smooth_train_loss": 2.9945010887296344, - "total_training_time": 12091.669941663742, - "stage_training_flops": 2092926498766848000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2092926498766848000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004995.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004995.json deleted file mode 100644 index 1d5c5a63c9a993fee6487d36ef8202a0eff1b594..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004995.json +++ /dev/null @@ -1,137 +0,0 @@ -{ - "step": 4995, - "experiment_id": "think-d12-r20-2epoch", - "val_bpb": 1.0624507774476128, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-2epoch", - "wandb_run_id": "520f145b", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,2epoch", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": 4995, - "target_flops": -1.0, - "target_param_data_ratio": -1.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", - "experiment_id": "think-d12-r20-2epoch", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", - "tokenizer_fingerprint": "90b338f6bf263273", - "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-2epoch", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-2epoch", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 22, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "epochs": 2, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core-metric-max-per-task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-2epoch", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "2epoch" - ] - }, - "config_fingerprint": "23f3b18820c5ee2b", - "artifact_path": "experiments/think-d12-r20-2epoch" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-2epoch", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "23f3b18820c5ee2b" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 0, - "pos": 320147, - "epoch": 3, - "pq_idx": 0, - "rg_idx": 320147 - }, - "loop_state": { - "min_val_bpb": 1.0624507774476128, - "smooth_train_loss": 2.90704286111097, - "total_training_time": 13423.92495751381, - "stage_training_flops": 2323148413631201280, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2323148413631201280 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_000500.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_000500.pt deleted file mode 100644 index 9408a39ed2f80cba1279c249f03b745875625265..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:08bb9da8ddaf2d78ebb0f6ba9103ca00e14d585b1afd6d0251b7424ab803c296 -size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_001000.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_001000.pt deleted file mode 100644 index 1155eba52a00d58b1eedb1507a93afd5a567c3e2..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2c90c13ad23843d84d2bf5b14aa6983fb0180799994454e4b37ae7cb75bb334d -size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_001500.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_001500.pt deleted file mode 100644 index 58c52e086c1f6351a9dc99e4b1da2a29a2600912..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1b0b4158ee911d17dd6fc2c6d0743995c6193fc9f141a88d4f108de16e115e46 -size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_002000.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_002000.pt deleted file mode 100644 index 1c83943878ecec0704350c6175c8349da164feb8..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8bfca87221770ccc082e2f2a92d3841c36d99caecaf1210658a300129f31215b -size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_002500.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_002500.pt deleted file mode 100644 index f1b7d8887039f92750b95c6d7a94721e6deaa067..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc24f494c1ef1a8b3f253abd5772cff42aeacfff5dfa69c01763b1ebcf79050e -size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_003000.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_003000.pt deleted file mode 100644 index fea89b5fbec67d7c85ef2985dc600a67c331069c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/model_003000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d2860ed10f079d0fcbc95f4cf69bcdd5a07a81da9feb3106a5d976db4ba93661 -size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_003500.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_003500.pt deleted file mode 100644 index 4a58e0fd9d8c70c7d416070bc6acd7a66c2a9fbc..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/model_003500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:fbd68738a488021b353a396a70b875818376bdeecdff3d04fff8216fca80bdcb -size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_004000.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_004000.pt deleted file mode 100644 index 2cb72bea035e3e8c3c48a777f34e6396cabfd5b8..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/model_004000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f132eea86799e79202ca6200a74a2b8805361f625032160594fea811e09298fd -size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_004500.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_004500.pt deleted file mode 100644 index 31a6826c9e5dc3d42e1f26796ff9c888ab424ad4..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/model_004500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:7f2d7eaf6cc11768364a4ff56a6c9059045893d1d45262dfc7151221f9e9cbb6 -size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_004995.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_004995.pt deleted file mode 100644 index 5135e092880134bea1debd1a1eabd06b7a157624..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/model_004995.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c022d1ba93c3884b7ea7a5a057c60795804eb31d05cac61d836a782516eeb134 -size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 92dbcccdef01bb5abbdf9b39146152645882f103..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2332fc532e05e4d2533a0894e1a5c756253970da580d7fb07346d00d9a55ddab -size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index cbcb8925fa8c1b8a4e12dfe01f32f37e09594a8e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a546f0abf80972d9beda2221729f73e466f60393d37a6a6c5d9b7cdd60bba3b5 -size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index d8aff4671dc04dbd516a0925ab42dfe678b225d1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b365410fa3b9f9eaa0c27af68055aadbebbf744ecb40b2544da19304e5882cda -size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index ec5e1c7d826f62948b9bd4941d81c49095e176a8..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:877e0f396df383cb9d9eb606759de51bdaa492a0d3555b20fff7ad147664e87a -size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_002500_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_002500_rank0.pt deleted file mode 100644 index e3068198ce4068101d1b3391a079f3a7f87e314c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:ce7df5dff6218eaf94685e547251c15cb8cda948d42b12c411ada56e8907d944 -size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_003000_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_003000_rank0.pt deleted file mode 100644 index ebf5b52b0a6d55b1f5a625503cec2060276bcd7c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_003000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d53a7bd9273fc9b2c99be5318a1cab115821a82297b554835d666f4692c7599a -size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_003500_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_003500_rank0.pt deleted file mode 100644 index d70e72e7c1a7dd8a8da39fcd413c78e8c4cec2ad..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_003500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3e072400b0c09eeb8a00013d3f8736b0699095a8e9ddcf3f2a7cd9f918102dc8 -size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004000_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004000_rank0.pt deleted file mode 100644 index 2378b741f417de7b607ab6980b706a16a8e657b6..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:276f0d9e52c9854fc516b403d3a171a3ac69b52dd885968b2953312f8e3f7a5b -size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004500_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004500_rank0.pt deleted file mode 100644 index ad303ec33098e173f9f583324826f6a83fb03e83..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b498b7f2e96dec7de1a2910c2677001d22102a87edbfb35f4f18d852bdeb41d8 -size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004995_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004995_rank0.pt deleted file mode 100644 index 8b03797d50a7e74193a0be67177052024b09516e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004995_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:289b156bc47fff9efd4b34a7fbb750ce5e27e58ce10f9d0945ad2d4850e41eb3 -size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/config.json b/experiments/think-d12-r20-2epoch/config.json deleted file mode 100644 index 38496e2e4425511b765de1c223404ef12a957f2b..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/config.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-2epoch", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 22, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "epochs": 2, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": 500, - "core-metric-max-per-task": 50, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-2epoch", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "2epoch" - ] - }, - "config_fingerprint": "23f3b18820c5ee2b", - "artifact_path": "experiments/think-d12-r20-2epoch" -} diff --git a/experiments/think-d12-r20-2epoch/evals/core.json b/experiments/think-d12-r20-2epoch/evals/core.json deleted file mode 100644 index 63669f7323ced0eb6673a33d01485e0645ee48ba..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 4995)", - "step": 4995, - "bpb": {}, - "core_metric": 0.08720366535348274, - "core_results": { - "hellaswag_zeroshot": 0.2848038077354431, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.12346833199262619, - "arc_easy": 0.32281145453453064, - "arc_challenge": 0.20904436707496643, - "copa": 0.5600000023841858, - "commonsense_qa": 0.3054873049259186, - "piqa": 0.547878086566925, - "openbook_qa": 0.24800001084804535, - "lambada_openai": 0.24063651263713837, - "hellaswag": 0.28659629821777344, - "winograd": 0.5677655935287476, - "winogrande": 0.5011838674545288, - "bigbench_dyck_languages": 0.10900000482797623, - "agi_eval_lsat_ar": 0.27391302585601807, - "bigbench_cs_algorithms": 0.4386363625526428, - "bigbench_operators": 0.06190476566553116, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.02866603620350361, - "coqa": 0.062132030725479126, - "boolq": 0.6067278385162354, - "bigbench_language_identification": 0.2505999803543091 - }, - "centered_results": { - "hellaswag_zeroshot": 0.04640507698059082, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.12346833199262619, - "arc_easy": 0.09708193937937419, - "arc_challenge": -0.054607510566711426, - "copa": 0.12000000476837158, - "commonsense_qa": 0.1318591311573982, - "piqa": 0.0957561731338501, - "openbook_qa": -0.002666652202606201, - "lambada_openai": 0.24063651263713837, - "hellaswag": 0.048795064290364586, - "winograd": 0.13553118705749512, - "winogrande": 0.002367734909057617, - "bigbench_dyck_languages": 0.10900000482797623, - "agi_eval_lsat_ar": 0.09239128232002257, - "bigbench_cs_algorithms": 0.4386363625526428, - "bigbench_operators": 0.06190476566553116, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.02866603620350361, - "coqa": 0.062132030725479126, - "boolq": -0.03492674074674906, - "bigbench_language_identification": 0.17557753614335433 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/evals/samples.json b/experiments/think-d12-r20-2epoch/evals/samples.json deleted file mode 100644 index 1a577941b0f1680c511cbf0cfc24d1a73b93ea77..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 4995)", - "step": 4995, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the kingdom of France, and the capital of the kingdom of France" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the gold of the sun, and the gold of the moon is the gold of" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Sunday. \n\nThe day of the Lord's resurrection is the day of the Lord" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the heat of the sun, and the heat of the sun is the heat of" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, which is the centre of the earth's orbit" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the red, and the white, and the blue, and the yellow, and" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the" - } - ], - "unconditioned_samples": [ - "<|bos|>auberstaden a topaz sick list they produce.\n\nREALISM \n\nWITH \n\nENGLAND 1666\n\n instructive SKELETON BURIPLOMATIC CORRESPONDENCE \n\nRepassed two eldest daughters belonging to one of the father's ex-officials in the same family, who died at an advanced age without having powerfully cultivated the talent necessary for their maintenance, These letters prove (were the value of our correspondence greater) how much encreased the readiness and ability of our soldiers and sailors, at this early age, to study mischief and plunder, and the necessity of effort to circumvent temptation, for we exercise discipline on certain beginnings of adventure which are not", - "<|bos|> Society chiefly composed of the most generous, disinterested, unchanging and neutral, one seeming to delight in the social advancement smoothly deferred, and another to endure that it shall in no case be free from thie drudgery which has before been tainted by corruption or illegality. ture.\n\nNeither of these views has been fully arrived at had the party been constituted wholly devoted to a particular church creed, according to the actual ideas which have been prevalent in at least one quarter of the country; and the design of the bill democracy has been to enrode a sort of '-lineal faction, with every attribute of a consurgent peer", - "<|bos|>wench, Morh'ot's Memid' vow, 102 Harright's Lane of London | Phila. \n\nEsopus Britann'. man are to his names, 37 the mistress of the temple, 140 of a . reconciliation beto all yet people, 26 of Ogil's wall, 27 of the site now in Heselt' pax 'mancient house', pet PLOT, AND, full blown, 45 house, 45 of Ruple, Angel's Island, No. 5, 44 \n\n36 said to happe, 46 to graci and '", - "<|bos|> INTERESTS OF BOYS. FOR YOUNG MEN BY A. B. MENDENHALL COFFEEWOOD, EDITOR OF JENNERS \n\n12353-58-1861.\n\nProfessor's\n\nCONTENTS A. B. MENDENHALL COFFEEWOOD'S MANUAL GYNNER SONS.. \n\n11 \n\nIntroduction \n\nCHARACTER \n\n12 \n\n\u2026\u2026\u2026... 12 \n\nINTRODUCTION 13 \n\nPRELIMINARY . \n\nTHE SOCIALIST NINETEENTH CENTURY A. B. MENDENHALL COFFEEWOOD'S MANUAL. \n\nInterest in the SOCIALIST MOVEMENT OF 1863 Twentieth Century EVIDENCE TRAFFIC SOCIALIST FEUDS ITS WILL Organized Social Democracy Conditions Gover", - "<|bos|>ma\n\nThey are good to see him, for all we have to drink on a plantation.\n\nSermon from Olivella in \"Dunmore's Soldier,\" \n\nTo the Servians of Richmond. \n\n(Exilles : one year.-LIFE and DEATH, 25 to 30.)\n\nHampton: January 25, 1776-7 \n\nBLOODY morning Our Thomas eat, nourish us with drink \n\nAnd to quench thirst we send him Seven days with Good\n\n bread and which he hath not drunk, we conceive it a very seasonable decertime to him, Which we may drink in with him on a", - "<|bos|>medyal Be. Jan. Sept. Feb. Esterhen Gebr.' (by Gesen.) \n\nMrs. he has blessed Be. day.' Anch., Seventy1; so is he was on shore.\n\nCHAP. XXVIII. \n\nAge as ordinarily blest;-the heavenly Monarchy; an There be gifted of the God of angels! It comes to the consciousness of Divine prerogative that this utterance enables Milton to encourage his inspirers to plunge into the vortex of this world of doubt and fears; to learn from it others may become irksome as he shrinks into himself, without The highest possible improvement afterwards", - "<|bos|>for Freeman. \n\n61. Meintum\n\nSmall, Articles called on the title and Acts again.\n\nActs by the Acts of 7th\n\nCont 1861-1918. \n\nApoplexies. Mystery was a much discussed matter in most days of munificence, and the request for compulsion among officials was leading.\n\nLetters Mat. xxii., xxvii., xxviii., xxxii., xxxiii., xxxiii., as well as ae letters in the Edward treaty of alliance were not called for in the The Iron Crusader, but the Mount Vernon consular offer of amity which these Nath. xxii", - "<|bos|> followed steamboats common to the three countries, this country and Russia, with the Metropolitan Districts of Manitoba, Galway, Ulster, and Connaught. The whole province is possessed of three great magazines for small ships, which are named the Minecigs. St Joseph at Prospect Hill, Kanssens at Harecastle, Stoketa and the Orinoco the other governors receiving the government of the province. \n\nThe principal city of the province is Noul. Every thing went on very comfortably till this great war which we call the present war, although our national disasters rose, owing to our" - ] -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/evals/val_bpb.json b/experiments/think-d12-r20-2epoch/evals/val_bpb.json deleted file mode 100644 index 0779b3e88b1fdeb6f5749b00c704cf45fa7e76b3..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 4995)", - "step": 4995, - "bpb": { - "val": 1.0107521146719778 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/run.json b/experiments/think-d12-r20-2epoch/run.json deleted file mode 100644 index 3285685ff027cf001eb6ed8b8ff499bae16aea77..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-r20-2epoch", - "stage": "base", - "base_experiment_id": "think-d12-r20-2epoch", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "95d46d80121d6088", - "wandb_run_id": "520f145b", - "created_at": 1781532282 -} diff --git a/experiments/think-d12-r20-2epoch/summary.json b/experiments/think-d12-r20-2epoch/summary.json deleted file mode 100644 index 7ff5139d1291229f45175d4e01694851eed495c4..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "think-d12-r20-2epoch", - "stage": "base", - "base_experiment_id": "think-d12-r20-2epoch", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset", - "dataset_revision": "main", - "step": 4995, - "depth": 12, - "target_param_data_ratio": null, - "training_tokens": 2618818560, - "final_sampled_val_bpb": 1.0624507774476128, - "minimum_sampled_val_bpb": 1.0624507774476128, - "full_val_bpb": 1.0107521146719778, - "core_metric": 0.08720366535348274, - "centered_results": { - "hellaswag_zeroshot": 0.04640507698059082, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.12346833199262619, - "arc_easy": 0.09708193937937419, - "arc_challenge": -0.054607510566711426, - "copa": 0.12000000476837158, - "commonsense_qa": 0.1318591311573982, - "piqa": 0.0957561731338501, - "openbook_qa": -0.002666652202606201, - "lambada_openai": 0.24063651263713837, - "hellaswag": 0.048795064290364586, - "winograd": 0.13553118705749512, - "winogrande": 0.002367734909057617, - "bigbench_dyck_languages": 0.10900000482797623, - "agi_eval_lsat_ar": 0.09239128232002257, - "bigbench_cs_algorithms": 0.4386363625526428, - "bigbench_operators": 0.06190476566553116, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.02866603620350361, - "coqa": 0.062132030725479126, - "boolq": -0.03492674074674906, - "bigbench_language_identification": 0.17557753614335433 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the kingdom of France, and the capital of the kingdom of France" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the gold of the sun, and the gold of the moon is the gold of" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Sunday. \n\nThe day of the Lord's resurrection is the day of the Lord" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the heat of the sun, and the heat of the sun is the heat of" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, which is the centre of the earth's orbit" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the red, and the white, and the blue, and the yellow, and" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the" - } - ], - "unconditioned_samples": [ - "<|bos|>auberstaden a topaz sick list they produce.\n\nREALISM \n\nWITH \n\nENGLAND 1666\n\n instructive SKELETON BURIPLOMATIC CORRESPONDENCE \n\nRepassed two eldest daughters belonging to one of the father's ex-officials in the same family, who died at an advanced age without having powerfully cultivated the talent necessary for their maintenance, These letters prove (were the value of our correspondence greater) how much encreased the readiness and ability of our soldiers and sailors, at this early age, to study mischief and plunder, and the necessity of effort to circumvent temptation, for we exercise discipline on certain beginnings of adventure which are not", - "<|bos|> Society chiefly composed of the most generous, disinterested, unchanging and neutral, one seeming to delight in the social advancement smoothly deferred, and another to endure that it shall in no case be free from thie drudgery which has before been tainted by corruption or illegality. ture.\n\nNeither of these views has been fully arrived at had the party been constituted wholly devoted to a particular church creed, according to the actual ideas which have been prevalent in at least one quarter of the country; and the design of the bill democracy has been to enrode a sort of '-lineal faction, with every attribute of a consurgent peer", - "<|bos|>wench, Morh'ot's Memid' vow, 102 Harright's Lane of London | Phila. \n\nEsopus Britann'. man are to his names, 37 the mistress of the temple, 140 of a . reconciliation beto all yet people, 26 of Ogil's wall, 27 of the site now in Heselt' pax 'mancient house', pet PLOT, AND, full blown, 45 house, 45 of Ruple, Angel's Island, No. 5, 44 \n\n36 said to happe, 46 to graci and '", - "<|bos|> INTERESTS OF BOYS. FOR YOUNG MEN BY A. B. MENDENHALL COFFEEWOOD, EDITOR OF JENNERS \n\n12353-58-1861.\n\nProfessor's\n\nCONTENTS A. B. MENDENHALL COFFEEWOOD'S MANUAL GYNNER SONS.. \n\n11 \n\nIntroduction \n\nCHARACTER \n\n12 \n\n\u2026\u2026\u2026... 12 \n\nINTRODUCTION 13 \n\nPRELIMINARY . \n\nTHE SOCIALIST NINETEENTH CENTURY A. B. MENDENHALL COFFEEWOOD'S MANUAL. \n\nInterest in the SOCIALIST MOVEMENT OF 1863 Twentieth Century EVIDENCE TRAFFIC SOCIALIST FEUDS ITS WILL Organized Social Democracy Conditions Gover", - "<|bos|>ma\n\nThey are good to see him, for all we have to drink on a plantation.\n\nSermon from Olivella in \"Dunmore's Soldier,\" \n\nTo the Servians of Richmond. \n\n(Exilles : one year.-LIFE and DEATH, 25 to 30.)\n\nHampton: January 25, 1776-7 \n\nBLOODY morning Our Thomas eat, nourish us with drink \n\nAnd to quench thirst we send him Seven days with Good\n\n bread and which he hath not drunk, we conceive it a very seasonable decertime to him, Which we may drink in with him on a", - "<|bos|>medyal Be. Jan. Sept. Feb. Esterhen Gebr.' (by Gesen.) \n\nMrs. he has blessed Be. day.' Anch., Seventy1; so is he was on shore.\n\nCHAP. XXVIII. \n\nAge as ordinarily blest;-the heavenly Monarchy; an There be gifted of the God of angels! It comes to the consciousness of Divine prerogative that this utterance enables Milton to encourage his inspirers to plunge into the vortex of this world of doubt and fears; to learn from it others may become irksome as he shrinks into himself, without The highest possible improvement afterwards", - "<|bos|>for Freeman. \n\n61. Meintum\n\nSmall, Articles called on the title and Acts again.\n\nActs by the Acts of 7th\n\nCont 1861-1918. \n\nApoplexies. Mystery was a much discussed matter in most days of munificence, and the request for compulsion among officials was leading.\n\nLetters Mat. xxii., xxvii., xxviii., xxxii., xxxiii., xxxiii., as well as ae letters in the Edward treaty of alliance were not called for in the The Iron Crusader, but the Mount Vernon consular offer of amity which these Nath. xxii", - "<|bos|> followed steamboats common to the three countries, this country and Russia, with the Metropolitan Districts of Manitoba, Galway, Ulster, and Connaught. The whole province is possessed of three great magazines for small ships, which are named the Minecigs. St Joseph at Prospect Hill, Kanssens at Harecastle, Stoketa and the Orinoco the other governors receiving the government of the province. \n\nThe principal city of the province is Noul. Every thing went on very comfortably till this great war which we call the present war, although our national disasters rose, owing to our" - ], - "training_time_seconds": 13423.92495751381, - "stage_training_flops": 2.3231484136312013e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 2.3231484136312013e+18, - "config_fingerprint": "23f3b18820c5ee2b", - "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/520f145b", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r20-2epoch", - "dataset_fingerprint": "f743a21ad2774428", - "tokenizer_fingerprint": "90b338f6bf263273", - "unique_train_tokens": 1309305551, - "effective_epochs": 2.0001584488814252 -} diff --git a/experiments/think-d12-r20-2epoch/tokenizer/experiment_tokenizer.json b/experiments/think-d12-r20-2epoch/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 839cfa5fffe02d709924f344017ef419dae7b3ab..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "think-d12-r20-2epoch", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 22, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1781534280 -} diff --git a/experiments/think-d12-r20-2epoch/tokenizer/token_bytes.pt b/experiments/think-d12-r20-2epoch/tokenizer/token_bytes.pt deleted file mode 100644 index 57df1c9e6f21a9cbfcb1131d0a730155c6eea7e1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:dde6733ef165ec5a1c250285a2f8bf7cf554039b0c23cbe98de70056c30d7901 -size 132649 diff --git a/experiments/think-d12-r20-2epoch/tokenizer/tokenizer.pkl b/experiments/think-d12-r20-2epoch/tokenizer/tokenizer.pkl deleted file mode 100644 index bebc9e00ea38cd78af03294576437cf234f34383..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-2epoch/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:edf4a58d6a4778da458e66a7e10b61ba23d78c27404807aa4fb5e057a0ec964d -size 404047 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_000500.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_000500.json deleted file mode 100644 index c518526b52cefee7b83076a11ef97683fdaa6350..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 500, - "experiment_id": "think-d12-r20-alt44", - "val_bpb": 1.3434480934243915, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-alt44", - "wandb_run_id": "4b198ee6", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,alt44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", - "experiment_id": "think-d12-r20-alt44", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", - "tokenizer_fingerprint": "3dc1a109161e9100", - "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-alt44", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-alt44", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "min_train_shard": 44, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-alt44", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "alt44" - ] - }, - "config_fingerprint": "60b9890a3df95e25", - "artifact_path": "experiments/think-d12-r20-alt44" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-alt44", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "60b9890a3df95e25" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3434480934243915, - "smooth_train_loss": 3.6726269837043755, - "total_training_time": 1307.7268741130829, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_001000.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_001000.json deleted file mode 100644 index af0705706a66e98c3b5e72c0d71b3316056c4a44..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 1000, - "experiment_id": "think-d12-r20-alt44", - "val_bpb": 1.2710118912993729, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-alt44", - "wandb_run_id": "4b198ee6", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,alt44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", - "experiment_id": "think-d12-r20-alt44", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", - "tokenizer_fingerprint": "3dc1a109161e9100", - "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-alt44", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-alt44", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "min_train_shard": 44, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-alt44", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "alt44" - ] - }, - "config_fingerprint": "60b9890a3df95e25", - "artifact_path": "experiments/think-d12-r20-alt44" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-alt44", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "60b9890a3df95e25" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2710118912993729, - "smooth_train_loss": 3.5156419568827517, - "total_training_time": 2644.638623714447, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_001500.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_001500.json deleted file mode 100644 index e2456b76dc3c9a6a2913dd6c6342a09e4be5a783..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 1500, - "experiment_id": "think-d12-r20-alt44", - "val_bpb": 1.2377739712547688, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-alt44", - "wandb_run_id": "4b198ee6", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,alt44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", - "experiment_id": "think-d12-r20-alt44", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", - "tokenizer_fingerprint": "3dc1a109161e9100", - "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-alt44", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-alt44", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "min_train_shard": 44, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-alt44", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "alt44" - ] - }, - "config_fingerprint": "60b9890a3df95e25", - "artifact_path": "experiments/think-d12-r20-alt44" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-alt44", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "60b9890a3df95e25" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.2377739712547688, - "smooth_train_loss": 3.3787051177159433, - "total_training_time": 3981.4400374889374, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_002000.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_002000.json deleted file mode 100644 index 9ca4283cbd9091b3bcd3e7c1f3ebf0eec8440dea..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 2000, - "experiment_id": "think-d12-r20-alt44", - "val_bpb": 1.192725198125875, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-alt44", - "wandb_run_id": "4b198ee6", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,alt44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", - "experiment_id": "think-d12-r20-alt44", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", - "tokenizer_fingerprint": "3dc1a109161e9100", - "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-alt44", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-alt44", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "min_train_shard": 44, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-alt44", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "alt44" - ] - }, - "config_fingerprint": "60b9890a3df95e25", - "artifact_path": "experiments/think-d12-r20-alt44" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-alt44", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "60b9890a3df95e25" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.192725198125875, - "smooth_train_loss": 3.3640745998451362, - "total_training_time": 5317.553530216217, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_002500.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_002500.json deleted file mode 100644 index d3234eb93f0269c6c91baec2205d57c940219265..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/meta_002500.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 2500, - "experiment_id": "think-d12-r20-alt44", - "val_bpb": 1.1596344082826722, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-alt44", - "wandb_run_id": "4b198ee6", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,alt44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", - "experiment_id": "think-d12-r20-alt44", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", - "tokenizer_fingerprint": "3dc1a109161e9100", - "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-alt44", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-alt44", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "min_train_shard": 44, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-alt44", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "alt44" - ] - }, - "config_fingerprint": "60b9890a3df95e25", - "artifact_path": "experiments/think-d12-r20-alt44" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-alt44", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "60b9890a3df95e25" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 13, - "pos": 10792769, - "epoch": 1, - "pq_idx": 13, - "rg_idx": 10792769 - }, - "loop_state": { - "min_val_bpb": 1.1596344082826722, - "smooth_train_loss": 3.2646331317328183, - "total_training_time": 6654.347955942154, - "stage_training_flops": 1162736943759360000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1162736943759360000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_003000.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_003000.json deleted file mode 100644 index aa445d6b55352e691fd2207c0e80c1d9e027ab3c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/meta_003000.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 3000, - "experiment_id": "think-d12-r20-alt44", - "val_bpb": 1.1237689589285966, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-alt44", - "wandb_run_id": "4b198ee6", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,alt44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", - "experiment_id": "think-d12-r20-alt44", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", - "tokenizer_fingerprint": "3dc1a109161e9100", - "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-alt44", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-alt44", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "min_train_shard": 44, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-alt44", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "alt44" - ] - }, - "config_fingerprint": "60b9890a3df95e25", - "artifact_path": "experiments/think-d12-r20-alt44" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-alt44", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "60b9890a3df95e25" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 15, - "pos": 72944769, - "epoch": 1, - "pq_idx": 15, - "rg_idx": 72944769 - }, - "loop_state": { - "min_val_bpb": 1.1237689589285966, - "smooth_train_loss": 3.2428728970036347, - "total_training_time": 7990.569368600845, - "stage_training_flops": 1395284332511232000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1395284332511232000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_003500.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_003500.json deleted file mode 100644 index fda57c5444edc0f074c9bedb2a28edab2d912538..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/meta_003500.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 3500, - "experiment_id": "think-d12-r20-alt44", - "val_bpb": 1.1019667576210812, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-alt44", - "wandb_run_id": "4b198ee6", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,alt44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", - "experiment_id": "think-d12-r20-alt44", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", - "tokenizer_fingerprint": "3dc1a109161e9100", - "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-alt44", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-alt44", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "min_train_shard": 44, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-alt44", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "alt44" - ] - }, - "config_fingerprint": "60b9890a3df95e25", - "artifact_path": "experiments/think-d12-r20-alt44" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-alt44", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "60b9890a3df95e25" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 18, - "pos": 35096769, - "epoch": 1, - "pq_idx": 18, - "rg_idx": 35096769 - }, - "loop_state": { - "min_val_bpb": 1.1019667576210812, - "smooth_train_loss": 3.1139676059158647, - "total_training_time": 9326.703073263168, - "stage_training_flops": 1627831721263104000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1627831721263104000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_004000.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_004000.json deleted file mode 100644 index 9f0591d0eda955c67b4b54b6419b7bc1358cce86..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/meta_004000.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 4000, - "experiment_id": "think-d12-r20-alt44", - "val_bpb": 1.0801314281850647, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-alt44", - "wandb_run_id": "4b198ee6", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,alt44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", - "experiment_id": "think-d12-r20-alt44", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", - "tokenizer_fingerprint": "3dc1a109161e9100", - "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-alt44", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-alt44", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "min_train_shard": 44, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-alt44", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "alt44" - ] - }, - "config_fingerprint": "60b9890a3df95e25", - "artifact_path": "experiments/think-d12-r20-alt44" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-alt44", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "60b9890a3df95e25" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 20, - "pos": 97248769, - "epoch": 1, - "pq_idx": 20, - "rg_idx": 97248769 - }, - "loop_state": { - "min_val_bpb": 1.0801314281850647, - "smooth_train_loss": 2.9255487986998965, - "total_training_time": 10663.225820541382, - "stage_training_flops": 1860379110014976000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1860379110014976000 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_004200.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_004200.json deleted file mode 100644 index e1b25dcb05e3f75b17da43f5e55ed0fa0d3294a2..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/meta_004200.json +++ /dev/null @@ -1,138 +0,0 @@ -{ - "step": 4200, - "experiment_id": "think-d12-r20-alt44", - "val_bpb": 1.0739757134682362, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-alt44", - "wandb_run_id": "4b198ee6", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset,d12,ratio20,alt44", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", - "experiment_id": "think-d12-r20-alt44", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", - "tokenizer_fingerprint": "3dc1a109161e9100", - "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "think-d12-r20-alt44", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-alt44", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "min_train_shard": 44, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-alt44", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "alt44" - ] - }, - "config_fingerprint": "60b9890a3df95e25", - "artifact_path": "experiments/think-d12-r20-alt44" - }, - "stage": "base", - "base_experiment_id": "think-d12-r20-alt44", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "60b9890a3df95e25" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 22, - "pos": 2109569, - "epoch": 1, - "pq_idx": 22, - "rg_idx": 2109569 - }, - "loop_state": { - "min_val_bpb": 1.0739757134682362, - "smooth_train_loss": 2.961268984708144, - "total_training_time": 11197.636986970901, - "stage_training_flops": 1953398065515724800, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1953398065515724800 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_000500.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_000500.pt deleted file mode 100644 index d5466ef7f260b658cffd4b21f62676e679e4d8fb..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bc338f9658612bb2c61bcb0c3abf43b11fa1c77e1d4cd2a2cc6d2344d698eff7 -size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_001000.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_001000.pt deleted file mode 100644 index 9cfe46772eb93c556f4bc704e081e86c2337951e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:daddc52f2e1f2bfc66a3667de7930a609b1cf5a961f3c43e70eb4ecbc1bf66f2 -size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_001500.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_001500.pt deleted file mode 100644 index 95cc350408dac5c2906e9fdc8a053009995ce851..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:89de4208cd428992f1843c9573b2d265eb626aed058822e1bf67e31ace61e68a -size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_002000.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_002000.pt deleted file mode 100644 index 8f5841b8531850042cfe1a7aa7d86ac166ebd2c5..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8e7dba25d5d192e349efe182eeb2d9e3869da546db073e0f08ec206c60af4cf1 -size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_002500.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_002500.pt deleted file mode 100644 index bd321fb95d5aa946677b94b5e1b182b21075c56d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3c534b557b77646330a8ccea71f1ed5be71a26177f424604bd294389290ae13c -size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_003000.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_003000.pt deleted file mode 100644 index b4e33ee84a1952fb549830709f963842fb7a5a2c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/model_003000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:20cdd57998ef483d11efc442342d56cbcd6f2b1bfa2f2d94d720495126f1ad26 -size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_003500.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_003500.pt deleted file mode 100644 index c574a2b464f72abdd6ea0ca90dfb42f7c14b3390..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/model_003500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:ee39c24b9abf8cb1fd56ecd724c9c75a2234317549efd608cf18a5ea7e08f867 -size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_004000.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_004000.pt deleted file mode 100644 index 3260d82ba2cd56a41f45ae4b89460346f2bdb77a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/model_004000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9e2e3318b7c936109b78275bd929933c8b641b4d627cb8d61ceb6a15fb1d6696 -size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_004200.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_004200.pt deleted file mode 100644 index f4fcff7fe4833a12f96b964c65cf86cd83908d7b..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/model_004200.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b67a3866f260b405a9a17a766c684c22fcd5bacd68ac9fc5e4a2b9706c3672f0 -size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 9d0cf69810eaf680e28f4d6f8accc07d16f2fe88..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:83686f19e2a36223f557c5b809a0a4dc6b26ed96d9f3fec1998ad7feb6d83c81 -size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 97093521ce8bd53d1bbad8121aa98de585ca2a51..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:cfd90df1f6664d71a8b3804403404bd17bb67624aea05960fcbfc80d457dc225 -size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index d66e6a3f7aee939726dd46616f40a949835fad9c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4cfb2381d2c19cd5685c4b568ee805a54cf53503606533bd5d0fab11689cb661 -size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 143c7d3ce6f9b0ddf9a00922fb860a706c54e930..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:ca2c999553741ab8b6ab1e9243d633e42c02a77d5d919699180673f7362fc1d8 -size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_002500_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_002500_rank0.pt deleted file mode 100644 index 6f74faaa86f3ddad97d1da35972a69e2b648d0ba..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2ecc710a1ad4a3e4840c3b856bc1c97b524de991db5e143728ac1699946ea783 -size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_003000_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_003000_rank0.pt deleted file mode 100644 index b34099279b728ed8d7b40d68c75515ebd92709de..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/optim_003000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:93ccf45bb18ce1a8a28de29d2367789a62c4ea00dd178af88e88c6cf1c751bab -size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_003500_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_003500_rank0.pt deleted file mode 100644 index cc1ea143bf9e07227506a340e7b3d23012138b75..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/optim_003500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bef09d5f4458358e4f0c5fc3c591d14f6fcc48cedcac6fb5d6ba28c8576024bc -size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_004000_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_004000_rank0.pt deleted file mode 100644 index 864c9995e4fecbdf3b77b65fe379fb85d1e5ac1d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/optim_004000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:273102ab4d2ef106bf25091f57305ebac76387952177e5fe9019b89c08f206bd -size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_004200_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_004200_rank0.pt deleted file mode 100644 index 0b789be8e232ccf18e6adf52baf62252a6520e6d..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/base_checkpoints/optim_004200_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6ab30b78fc55f66d74ad106c4df7fbca65abf43b7cf1c8f52ba8b096500d1ce5 -size 1246165357 diff --git a/experiments/think-d12-r20-alt44/config.json b/experiments/think-d12-r20-alt44/config.json deleted file mode 100644 index a7094f72d48473b7398602d324ccae11d0490929..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/config.json +++ /dev/null @@ -1,57 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20-alt44", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "min_train_shard": 44, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20-alt44", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20", - "alt44" - ] - }, - "config_fingerprint": "60b9890a3df95e25", - "artifact_path": "experiments/think-d12-r20-alt44" -} diff --git a/experiments/think-d12-r20-alt44/evals/core.json b/experiments/think-d12-r20-alt44/evals/core.json deleted file mode 100644 index eed98247e664236d9fd0b67615d50dae8d0788d3..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 4200)", - "step": 4200, - "bpb": {}, - "core_metric": 0.0774704140744213, - "core_results": { - "hellaswag_zeroshot": 0.2824138402938843, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.057133015245199203, - "arc_easy": 0.32407405972480774, - "arc_challenge": 0.22098974883556366, - "copa": 0.5299999713897705, - "commonsense_qa": 0.3030303120613098, - "piqa": 0.5516865849494934, - "openbook_qa": 0.2460000067949295, - "lambada_openai": 0.22084222733974457, - "hellaswag": 0.2816171944141388, - "winograd": 0.5457875728607178, - "winogrande": 0.4956590235233307, - "bigbench_dyck_languages": 0.11100000888109207, - "agi_eval_lsat_ar": 0.2652173936367035, - "bigbench_cs_algorithms": 0.38181817531585693, - "bigbench_operators": 0.10952381044626236, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.03405865654349327, - "coqa": 0.0860578715801239, - "boolq": 0.593883752822876, - "bigbench_language_identification": 0.25049999356269836 - }, - "centered_results": { - "hellaswag_zeroshot": 0.04321845372517904, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.057133015245199203, - "arc_easy": 0.09876541296641032, - "arc_challenge": -0.03868033488591512, - "copa": 0.059999942779541016, - "commonsense_qa": 0.12878789007663724, - "piqa": 0.10337316989898682, - "openbook_qa": -0.005333324273427327, - "lambada_openai": 0.22084222733974457, - "hellaswag": 0.04215625921885172, - "winograd": 0.09157514572143555, - "winogrande": -0.008681952953338623, - "bigbench_dyck_languages": 0.11100000888109207, - "agi_eval_lsat_ar": 0.08152174204587935, - "bigbench_cs_algorithms": 0.38181817531585693, - "bigbench_operators": 0.10952381044626236, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.03405865654349327, - "coqa": 0.0860578715801239, - "boolq": -0.06872696625558952, - "bigbench_language_identification": 0.1754675396729355 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/evals/samples.json b/experiments/think-d12-r20-alt44/evals/samples.json deleted file mode 100644 index 84b89fcbe14b41eea76fb2698f0cde99a3147e5c..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 4200)", - "step": 4200, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is 10,000,000 francs, or 10," - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as the symbol of the sun, and the same as the symbol of" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nIf yesterday was Saturday, then tomorrow will be Sunday. \n\n" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is a very common error, and is not a little remarkable. It is a very" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, 2. The moon, 3. The" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a reddish brown, and the color of the skin is a redd" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the square of the number of the square of the number of the" - } - ], - "unconditioned_samples": [ - "<|bos|>asi that of her reputed father. There were in her last year or two here so many words of reproach, accusing all her priests, waiting on her travail of life, that they called her true wife therefore, and implored for forgiveness. There was one again who, after so many curses, murmured induced hope through advantages, confidence derived by human hopes, thus she bethought her of the tempted profligate, bethinking her what she was doing; and there was a third person who, in attempting to induce her to accept either the fall or the punishment of the false accuser, vindicated herself from the foul", - "<|bos|> Carrapal, the lips visible, and, unlike \n\nLeblanc, its lips sinuous, and\n\n Rod consumed them and, what is more, they are ashes of poisons, and, to shut out therefrom the \n\n6 18k of a retail price, every part of his body has an appletree within his bowels, which they are said to destroy; his heart acts fiery compositions. \n\nRABINS. A narrow process on a chemical tree. \n\n1\u00ba 1008. GOLD AND SILVER BENCH. \n\n1179. BLACK and red ink, one in dye of black or black t", - "<|bos|>ANA COMMISSIONER'S EXPRESS-POWER OF STUDIES \n\nWants not to be shown in museum rulings without such warrant, as where a survey is not admissible as error to show the boundary of a\n\nKANSAS HISTORY \n\nKENTUCKY OPINIONS OR DOCUMENTS 23 \n\nREPORTS \n\nCOMPL \n\nJOHNSTON \n\nJOHNSON \n\nMORE \n\nJOHN GREEN \n\nST. LOUIS \n\nJOHNSTON \n\nTRAPP \n\nLOVE \n\nLOUISE LOAKE \n\n ANDORE \n\nChastity de- with was reported by the Superintendent of Education opinion that Washington was the nearest approximation Vincennes, the wisdom merchandise to do the actual route to to coalfields is", - "<|bos|> wield hard 160,6 G.C.M \n\n3.1\n\n10439 An Introduction \n\nFind a good Equinoxa (S.P.M.C.C.) \n\nFind a good\n\nEquinoxa \n\nNa (S.P.M.C.C.) \n\nAnt oc \"Soft may the light serene \n\nCome through to us that liveth \n\nSweet air of heaven, and while flutes \n\nBurn on the Sabbaths that thy soul \n\nBlows with unspeakable bliss\" (S.P.C.) Fra\n\n\u0399\u0399 Chant chosen rather than a plowbox) \n\n11 be the good gentle wind", - "<|bos|> Lac they are still four small animals which deserve to be treated as tame things.' \n\nFor making them wild I have also made in three mounds the beating of large pipes, a process in which Paul Ursula is interribted with a small fsel, or rather escape of the mauds: will,' says Augustine, 'only \n\n1 Hist. Ancient Cities, iii, 4.\n\nWAORIES IN BAB LANGUAGES \n\nmuch trouble.' For the range and torridity of the Mediterranean they which chiefly merit attention are these: (I.) the solitary growth of the baize trees thrown out by the Tartars from Aujib", - "<|bos|>PLmillan, 1,433), and p. 758. Ay, now tell me, but let him tell thee v.: III.\n\nI must apply it to touch the where he has desired to see it. It need hardly be said, that in the nursery, Diogenes seeing the anachronymus hesitates: or, in the pining out of Franklin's poisoned spider, passage \u00e4n his retort to the school-house, was saying: \"I cannot shave better than I saved from drowning, and knocked out again from living ten years, and rained down curds", - "<|bos|>1749.11 \n\n17434721S+ Vpels during 1749), and again at the end of the year 7. 9 1\n\n2044 020 035 844 \n\nStowe, 91", - "<|bos|> PEOPLE MEET OTHERS IN THEIT\u00c9 OF A GREAT BUNTER HILL \n\nTHE \n\nTHE READERS THAT FORTUNATELY MEET\n\nTWO OCCURRENCES NEAR THE TROAD NICOLO JAM THE BRITISH \n\nThe well-known type of New England ferry-boat presented to Captain John Chapman Wilson some years ago has been very generally appreciated (or ignored) and a current of eighty yards was paid down by Professor\n\nCleghorn as 54 cents. This was his generous way of paying anything to a schoolmate under any circumstances. When Capt. John Chapman first tried to reach the shore of St. Laurent in Camp Bellingham, he did not" - ] -} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/evals/val_bpb.json b/experiments/think-d12-r20-alt44/evals/val_bpb.json deleted file mode 100644 index ebeb5f9a060a85d65eaddf873258e61696678699..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 4200)", - "step": 4200, - "bpb": { - "val": 1.0180559919724057 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/run.json b/experiments/think-d12-r20-alt44/run.json deleted file mode 100644 index 33d70e05fbc50d65a397a92f05c178475878ef88..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-r20-alt44", - "stage": "base", - "base_experiment_id": "think-d12-r20-alt44", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "60b9890a3df95e25", - "wandb_run_id": "4b198ee6", - "created_at": 1781721383 -} diff --git a/experiments/think-d12-r20-alt44/summary.json b/experiments/think-d12-r20-alt44/summary.json deleted file mode 100644 index b6c93c2f8d686d561ff5747d6010390ef0e047f8..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "think-d12-r20-alt44", - "stage": "base", - "base_experiment_id": "think-d12-r20-alt44", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset", - "dataset_revision": "main", - "step": 4200, - "depth": 12, - "target_param_data_ratio": 20.0, - "training_tokens": 2202009600, - "final_sampled_val_bpb": 1.0739757134682362, - "minimum_sampled_val_bpb": 1.0739757134682362, - "full_val_bpb": 1.0180559919724057, - "core_metric": 0.0774704140744213, - "centered_results": { - "hellaswag_zeroshot": 0.04321845372517904, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.057133015245199203, - "arc_easy": 0.09876541296641032, - "arc_challenge": -0.03868033488591512, - "copa": 0.059999942779541016, - "commonsense_qa": 0.12878789007663724, - "piqa": 0.10337316989898682, - "openbook_qa": -0.005333324273427327, - "lambada_openai": 0.22084222733974457, - "hellaswag": 0.04215625921885172, - "winograd": 0.09157514572143555, - "winogrande": -0.008681952953338623, - "bigbench_dyck_languages": 0.11100000888109207, - "agi_eval_lsat_ar": 0.08152174204587935, - "bigbench_cs_algorithms": 0.38181817531585693, - "bigbench_operators": 0.10952381044626236, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.03405865654349327, - "coqa": 0.0860578715801239, - "boolq": -0.06872696625558952, - "bigbench_language_identification": 0.1754675396729355 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is 10,000,000 francs, or 10," - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as the symbol of the sun, and the same as the symbol of" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nIf yesterday was Saturday, then tomorrow will be Sunday. \n\n" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is a very common error, and is not a little remarkable. It is a very" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, 2. The moon, 3. The" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is a reddish brown, and the color of the skin is a redd" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the square of the number of the square of the number of the" - } - ], - "unconditioned_samples": [ - "<|bos|>asi that of her reputed father. There were in her last year or two here so many words of reproach, accusing all her priests, waiting on her travail of life, that they called her true wife therefore, and implored for forgiveness. There was one again who, after so many curses, murmured induced hope through advantages, confidence derived by human hopes, thus she bethought her of the tempted profligate, bethinking her what she was doing; and there was a third person who, in attempting to induce her to accept either the fall or the punishment of the false accuser, vindicated herself from the foul", - "<|bos|> Carrapal, the lips visible, and, unlike \n\nLeblanc, its lips sinuous, and\n\n Rod consumed them and, what is more, they are ashes of poisons, and, to shut out therefrom the \n\n6 18k of a retail price, every part of his body has an appletree within his bowels, which they are said to destroy; his heart acts fiery compositions. \n\nRABINS. A narrow process on a chemical tree. \n\n1\u00ba 1008. GOLD AND SILVER BENCH. \n\n1179. BLACK and red ink, one in dye of black or black t", - "<|bos|>ANA COMMISSIONER'S EXPRESS-POWER OF STUDIES \n\nWants not to be shown in museum rulings without such warrant, as where a survey is not admissible as error to show the boundary of a\n\nKANSAS HISTORY \n\nKENTUCKY OPINIONS OR DOCUMENTS 23 \n\nREPORTS \n\nCOMPL \n\nJOHNSTON \n\nJOHNSON \n\nMORE \n\nJOHN GREEN \n\nST. LOUIS \n\nJOHNSTON \n\nTRAPP \n\nLOVE \n\nLOUISE LOAKE \n\n ANDORE \n\nChastity de- with was reported by the Superintendent of Education opinion that Washington was the nearest approximation Vincennes, the wisdom merchandise to do the actual route to to coalfields is", - "<|bos|> wield hard 160,6 G.C.M \n\n3.1\n\n10439 An Introduction \n\nFind a good Equinoxa (S.P.M.C.C.) \n\nFind a good\n\nEquinoxa \n\nNa (S.P.M.C.C.) \n\nAnt oc \"Soft may the light serene \n\nCome through to us that liveth \n\nSweet air of heaven, and while flutes \n\nBurn on the Sabbaths that thy soul \n\nBlows with unspeakable bliss\" (S.P.C.) Fra\n\n\u0399\u0399 Chant chosen rather than a plowbox) \n\n11 be the good gentle wind", - "<|bos|> Lac they are still four small animals which deserve to be treated as tame things.' \n\nFor making them wild I have also made in three mounds the beating of large pipes, a process in which Paul Ursula is interribted with a small fsel, or rather escape of the mauds: will,' says Augustine, 'only \n\n1 Hist. Ancient Cities, iii, 4.\n\nWAORIES IN BAB LANGUAGES \n\nmuch trouble.' For the range and torridity of the Mediterranean they which chiefly merit attention are these: (I.) the solitary growth of the baize trees thrown out by the Tartars from Aujib", - "<|bos|>PLmillan, 1,433), and p. 758. Ay, now tell me, but let him tell thee v.: III.\n\nI must apply it to touch the where he has desired to see it. It need hardly be said, that in the nursery, Diogenes seeing the anachronymus hesitates: or, in the pining out of Franklin's poisoned spider, passage \u00e4n his retort to the school-house, was saying: \"I cannot shave better than I saved from drowning, and knocked out again from living ten years, and rained down curds", - "<|bos|>1749.11 \n\n17434721S+ Vpels during 1749), and again at the end of the year 7. 9 1\n\n2044 020 035 844 \n\nStowe, 91", - "<|bos|> PEOPLE MEET OTHERS IN THEIT\u00c9 OF A GREAT BUNTER HILL \n\nTHE \n\nTHE READERS THAT FORTUNATELY MEET\n\nTWO OCCURRENCES NEAR THE TROAD NICOLO JAM THE BRITISH \n\nThe well-known type of New England ferry-boat presented to Captain John Chapman Wilson some years ago has been very generally appreciated (or ignored) and a current of eighty yards was paid down by Professor\n\nCleghorn as 54 cents. This was his generous way of paying anything to a schoolmate under any circumstances. When Capt. John Chapman first tried to reach the shore of St. Laurent in Camp Bellingham, he did not" - ], - "training_time_seconds": 11197.636986970901, - "stage_training_flops": 1.9533980655157248e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.9533980655157248e+18, - "config_fingerprint": "60b9890a3df95e25", - "git_commit_sha": "0ea306100c54d5e0950d28c092ec5291ee71c247", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/4b198ee6", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r20-alt44", - "dataset_fingerprint": "147853d67cb3de71", - "tokenizer_fingerprint": "3dc1a109161e9100", - "unique_train_tokens": 2268069888, - "effective_epochs": 0.970873786407767 -} diff --git a/experiments/think-d12-r20-alt44/tokenizer/experiment_tokenizer.json b/experiments/think-d12-r20-alt44/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 1f277b2e7847d48764700614f968e2ebad2bd276..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,19 +0,0 @@ -{ - "experiment_id": "think-d12-r20-alt44", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "min_train_shard": 44, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1781721409 -} diff --git a/experiments/think-d12-r20-alt44/tokenizer/token_bytes.pt b/experiments/think-d12-r20-alt44/tokenizer/token_bytes.pt deleted file mode 100644 index d5505db6f555c8497724954a1dd310ecfcf52b3b..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:271a4064bad6183970392e829ff7a8fd25ba37a811b880fde5f5b6d58823a5a0 -size 132649 diff --git a/experiments/think-d12-r20-alt44/tokenizer/tokenizer.pkl b/experiments/think-d12-r20-alt44/tokenizer/tokenizer.pkl deleted file mode 100644 index bb8aaabb88ba0c6eb3d6efe0e26704772111316e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20-alt44/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b99e5e10fbe663212046c441fd4d584a19c1a5531e8ebea04f8e42238998a4e4 -size 403342 diff --git a/experiments/think-d12-r20/base_checkpoints/meta_000500.json b/experiments/think-d12-r20/base_checkpoints/meta_000500.json deleted file mode 100644 index ba79b4dc95de94cd09f87814f615475ece474c1e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 500, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.5891939434241213, - "total_training_time": 1312.64857006073 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_001000.json b/experiments/think-d12-r20/base_checkpoints/meta_001000.json deleted file mode 100644 index bbca0e8342d8f36e4a06b1933ce176471b508a07..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 1000, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.4868806829553485, - "total_training_time": 2654.1426842212677 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_001500.json b/experiments/think-d12-r20/base_checkpoints/meta_001500.json deleted file mode 100644 index 4d052d841af639e6ed386c3bf952cdef76107cbf..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 1500, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.410999880107203, - "total_training_time": 3995.465323448181 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_002000.json b/experiments/think-d12-r20/base_checkpoints/meta_002000.json deleted file mode 100644 index 8ccecb680019f8c6b8dd9c312d17235d307c95bf..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 2000, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.452807694528099, - "total_training_time": 5336.5537366867065 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_002500.json b/experiments/think-d12-r20/base_checkpoints/meta_002500.json deleted file mode 100644 index 96fc7e712e052057bcc504c98aa1891ccee8b9e2..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/meta_002500.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 2500, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 13, - "pos": 10792769, - "epoch": 1, - "pq_idx": 13, - "rg_idx": 10792769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.327396609246394, - "total_training_time": 6677.6987290382385 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_003000.json b/experiments/think-d12-r20/base_checkpoints/meta_003000.json deleted file mode 100644 index 7ea0f8f11e2301baef3d24fa3c60d0fa0a52f9de..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/meta_003000.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 3000, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 15, - "pos": 72944769, - "epoch": 1, - "pq_idx": 15, - "rg_idx": 72944769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.048554485031434, - "total_training_time": 8018.697687149048 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_003500.json b/experiments/think-d12-r20/base_checkpoints/meta_003500.json deleted file mode 100644 index ea7c74460fdec7160133b9cbc32e9e20470153ee..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/meta_003500.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 3500, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 18, - "pos": 35096769, - "epoch": 1, - "pq_idx": 18, - "rg_idx": 35096769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 3.0863692227626416, - "total_training_time": 9359.67183303833 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_004000.json b/experiments/think-d12-r20/base_checkpoints/meta_004000.json deleted file mode 100644 index e43944a04fee67ab2d03020bbd670e2b024bf741..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/meta_004000.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 4000, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 20, - "pos": 97248769, - "epoch": 1, - "pq_idx": 20, - "rg_idx": 97248769 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 2.7993042062101186, - "total_training_time": 10700.389838218689 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_004200.json b/experiments/think-d12-r20/base_checkpoints/meta_004200.json deleted file mode 100644 index edcf82bc7f5d7ab29272189b9203fe91c518bbe2..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/meta_004200.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "step": 4200, - "val_bpb": null, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "dummy", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 20.0, - "device_batch_size": 16, - "total_batch_size": -1, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "eval_every": -1, - "eval_tokens": 41943040, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "d12-ratio20" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 22, - "pos": 2109569, - "epoch": 1, - "pq_idx": 22, - "rg_idx": 2109569 - }, - "loop_state": { - "min_val_bpb": Infinity, - "smooth_train_loss": 2.8461822660242255, - "total_training_time": 11236.57096004486 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/model_000500.pt b/experiments/think-d12-r20/base_checkpoints/model_000500.pt deleted file mode 100644 index 97e4c71cc3328cef1fabf54e4a1b8ecf3e240e32..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:00e14e7ad4637b08c6e5f649ebc7de7ed0d9fcb39a293b7b088a13c2ebb549f8 -size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_001000.pt b/experiments/think-d12-r20/base_checkpoints/model_001000.pt deleted file mode 100644 index e2d60950935b1958573cb2de2aa7a12151bc1f16..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:01e2a9466841de94c8c3a654a257c842c1764d3ac4cd79cf478f9f2b24859e7c -size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_001500.pt b/experiments/think-d12-r20/base_checkpoints/model_001500.pt deleted file mode 100644 index 5bad12b7a4a0a3593d00a4ba4d742ef0bd7e5945..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:03fb5d4241142f86b69396fe200295929df1b4bcc2ba5d4cc464c9627cf0a530 -size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_002000.pt b/experiments/think-d12-r20/base_checkpoints/model_002000.pt deleted file mode 100644 index 283db3b4a5e606111e288f51ab830fceb5ea3fb3..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c4db8d9fa2b5fa4ca14f65dfd8d5356c691c5c8772e839555f0fec6f8d300269 -size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_002500.pt b/experiments/think-d12-r20/base_checkpoints/model_002500.pt deleted file mode 100644 index 766aefe9911a42ba0478ae0effee731d7e248fe1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/model_002500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:abb54590a71621be44e418b133a42088aed1d84872a9726e9aa4ac8b7e418788 -size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_003000.pt b/experiments/think-d12-r20/base_checkpoints/model_003000.pt deleted file mode 100644 index 95408589ebb4a00a04f89c866519d954a17a96db..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/model_003000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:0a96aa9105d455f741602d43d807dd0b2732823c9c8be3465717f7090aa469eb -size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_003500.pt b/experiments/think-d12-r20/base_checkpoints/model_003500.pt deleted file mode 100644 index aa7696886c6993d18f98632b2a12d2e93c89c890..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/model_003500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bab4a0741306ce5945c9a86dabd233081b836f0e1f814dd1d9363cc1b9cce950 -size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_004000.pt b/experiments/think-d12-r20/base_checkpoints/model_004000.pt deleted file mode 100644 index 195c00c1e8c9542b4c3a4237f783261f127da27b..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/model_004000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6c967e927215ebca6af3f49011d98fd8fa4487c55e78485271878e382b7293d5 -size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_004200.pt b/experiments/think-d12-r20/base_checkpoints/model_004200.pt deleted file mode 100644 index c201085206abc74184928a0dac513e531432f6cc..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/model_004200.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:1dfab1c19f9bbd3dd27ed859b70cfd242d5ebaef734985dbef78d4998493a6f6 -size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 3c6f2f796a2895cf54026c01b22c14a52177d24e..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:62b942e737b779c2dbb3b14246ffc951fb9dce816e10e7ada936b81093faaca7 -size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 0272991a5a4536ff457d456339de4d750c8cda24..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f5fe23783669dcb79dda4bc0a66e2c536d9d3055fbcaf07840662e2c5b8f6950 -size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 88a0adfe1b99ce52dcf0807bf3881371f4ed1184..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9dcce76cabc6cdd8083af5f064ae391d8b84fc96d2608ab6740fffc2b1e2f5f2 -size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index b11fbe743399c95fabbd0696c2852a595104d922..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4dc97c55d9aa7e22cd38dcfc298b42066aeda741daf7d1b508575ac62c0e32de -size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_002500_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_002500_rank0.pt deleted file mode 100644 index 37d285b746991053e40d3e97b32448203c24c826..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/optim_002500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2550badb9d208d9f2e51765bafd6789cc2f9ee8b6fd5880e25790d427deb9556 -size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_003000_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_003000_rank0.pt deleted file mode 100644 index b78dee7e866f9e7c539a9d6ca1e48264d7b2c2b5..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/optim_003000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8ea5ec19b0375980f64cbdbb9709d040a01e2f546ebdcc380246f8c69d26b242 -size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_003500_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_003500_rank0.pt deleted file mode 100644 index 26f63d83d8e08a5bf8c6b3139f798f5dabfb10aa..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/optim_003500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9465372371e5d248972fe855b83234cbf5fde5bcf208eb7e2faf9f51cf207def -size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_004000_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_004000_rank0.pt deleted file mode 100644 index 8ad1e5577fd1f8636688a15b9f932609943770e6..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/optim_004000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c27aa4c1fea51a61cd5aef3654780852040f59b347b9f18c7a9f29620664b677 -size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_004200_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_004200_rank0.pt deleted file mode 100644 index dcfacec3437fca29f63311b48f8f86a813607bcd..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/base_checkpoints/optim_004200_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3a71d1f755bf5e13a1bd7741a3a1bbc1d9807468b5190c4e2380b91e0116eee0 -size 1246165237 diff --git a/experiments/think-d12-r20/config.json b/experiments/think-d12-r20/config.json deleted file mode 100644 index ca4f4971460638cb8ce3e0797217d0dcd2781f49..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/config.json +++ /dev/null @@ -1,55 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "think-d12-r20", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 44, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 20.0, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "think-d12-r20", - "group": "think-d12", - "tags": [ - "think-dataset", - "d12", - "ratio20" - ] - }, - "config_fingerprint": "4236ec51a84df577", - "artifact_path": "experiments/think-d12-r20" -} diff --git a/experiments/think-d12-r20/evals/val_bpb.json b/experiments/think-d12-r20/evals/val_bpb.json deleted file mode 100644 index a3685444ee7259a3f6389b3947e141787e52031b..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 4200)", - "step": 4200, - "bpb": { - "val": 1.0597927744819549 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/think-d12-r20/run.json b/experiments/think-d12-r20/run.json deleted file mode 100644 index c1a7fefa17921b8f72b628a7bcc3d2ceb1842236..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/run.json +++ /dev/null @@ -1,6 +0,0 @@ -{ - "experiment_id": "think-d12-r20", - "stage": "base", - "wandb_run_id": null, - "migration_note": "Migrated from the pre-lineage repository layout." -} diff --git a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/meta_000015.json b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/meta_000015.json deleted file mode 100644 index 0ed9a3f691091d60e600b73aefd16fda1c178857..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/meta_000015.json +++ /dev/null @@ -1,102 +0,0 @@ -{ - "step": 15, - "val_bpb": 0.9148607229178501, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "think-d12-r20-pre1930-authentic", - "wandb_run_id": "f1457d9c", - "wandb_group": "think-d12", - "wandb_tags": "sft,pre1930,ratio20", - "device_type": "", - "model_tag": null, - "model_step": null, - "base_checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20/base_checkpoints", - "base_step": 4200, - "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints", - "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20/tokenizer", - "resume_from_step": null, - "experiment_id": "think-d12-r20-pre1930-authentic", - "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/config.json", - "parent_cumulative_flops": 1.95339809193984e+18, - "tokenizer_fingerprint": "6e592b9b323f98bf", - "git_commit_sha": "16a49c72218157b03f4bb238a0fdc23f0bca8b18", - "load_optimizer": 1, - "num_iterations": -1, - "max_seq_len": null, - "device_batch_size": 8, - "total_batch_size": null, - "embedding_lr": null, - "unembedding_lr": null, - "matrix_lr": null, - "init_lr_frac": 0.8, - "warmup_ratio": 0.0, - "warmdown_ratio": 0.5, - "final_lr_frac": 0.0, - "eval_every": -1, - "eval_tokens": 20971520, - "chatcore_every": -1, - "chatcore_max_cat": -1, - "chatcore_max_sample": 24, - "save_every": -1, - "recipe": "pre1930", - "pre1930_epochs": 5, - "mmlu_epochs": 3, - "gsm8k_epochs": 4, - "resolved_experiment_config": { - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-authentic", - "data": { - "recipe": "pre1930", - "pre1930_epochs": 5 - }, - "training": { - "num_iterations": -1, - "device_batch_size": 8, - "eval_every": -1, - "chatcore_every": -1, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "pre1930", - "ratio20" - ] - }, - "config_fingerprint": "d3378357cef17359", - "artifact_path": "experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic" - }, - "stage": "sft", - "base_experiment_id": null, - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "d3378357cef17359" - }, - "loop_state": { - "step": 15, - "total_training_time": 14.16254210472107, - "min_val_bpb": 0.9148607229178501, - "smooth_train_loss": 1.5855611060774228, - "mfu": 52.60094494392411, - "tok_per_sec": 185002, - "stage_training_flops": 6976421662556160.0, - "inherited_parent_flops": 1.95339809193984e+18, - "cumulative_pipeline_training_flops": 1.9603745136023962e+18 - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/model_000015.pt b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/model_000015.pt deleted file mode 100644 index dde9bb22e5578ccf650a6ac7c901146a455916af..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/model_000015.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4893fa6fd812a1b5ec282a174f3c7bd6df69ce5f73d785e59537e2378bc2c833 -size 792761690 diff --git a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/optim_000015_rank0.pt b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/optim_000015_rank0.pt deleted file mode 100644 index a91d44ccdb7da88396f21313673e1405d10feadb..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/optim_000015_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4d72645c73b58777d4cef8e50f304b632f4dd09f7bb023f985ba458c19669207 -size 1246165357 diff --git a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/config.json b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/config.json deleted file mode 100644 index a8d72cacc5c7fa2896dc5423493c9298441b22c1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/config.json +++ /dev/null @@ -1,32 +0,0 @@ -{ - "schema_version": 1, - "stage": "sft", - "experiment_suffix": "pre1930-authentic", - "data": { - "recipe": "pre1930", - "pre1930_epochs": 5 - }, - "training": { - "num_iterations": -1, - "device_batch_size": 8, - "eval_every": -1, - "chatcore_every": -1, - "save_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "enabled": true, - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "group": "think-d12", - "tags": [ - "sft", - "pre1930", - "ratio20" - ] - }, - "config_fingerprint": "d3378357cef17359", - "artifact_path": "experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic" -} diff --git a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/run.json b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/run.json deleted file mode 100644 index 8716944f8a2466c00be953dc4c6746b800d8b0b9..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "think-d12-r20-pre1930-authentic", - "stage": "sft", - "base_experiment_id": "think-d12-r20", - "parent_experiment_id": "think-d12-r20", - "parent_checkpoint_step": null, - "config_fingerprint": "d3378357cef17359", - "wandb_run_id": "f1457d9c", - "created_at": 1782154820 -} diff --git a/experiments/think-d12-r20/summary.json b/experiments/think-d12-r20/summary.json deleted file mode 100644 index 6e881a9514eb34f23dea96541f8e7559ef9441a4..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/summary.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "experiment_id": "think-d12-r20", - "stage": "base", - "base_experiment_id": "think-d12-r20", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset", - "dataset_revision": "main", - "step": 4200, - "depth": 12, - "target_param_data_ratio": 20.0, - "training_tokens": 2202009600, - "final_sampled_val_bpb": null, - "minimum_sampled_val_bpb": Infinity, - "full_val_bpb": 1.0597927744819549, - "core_metric": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [], - "training_time_seconds": 11236.57096004486, - "stage_training_flops": 1.95339809193984e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.95339809193984e+18, - "config_fingerprint": "4236ec51a84df577", - "git_commit_sha": "44d6e64aec8e6a1ec2f2a621ca1b44574b705d66", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/None", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r20", - "dataset_fingerprint": "a6e1b3a100e0d8b3", - "tokenizer_fingerprint": "6e592b9b323f98bf" -} diff --git a/experiments/think-d12-r20/tokenizer/think_dataset_tokenizer.json b/experiments/think-d12-r20/tokenizer/think_dataset_tokenizer.json deleted file mode 100644 index fea3141f8940ad53370916cd7a3dc8743edbf3ad..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/tokenizer/think_dataset_tokenizer.json +++ /dev/null @@ -1,20 +0,0 @@ -{ - "dataset_repo": "jbduran/think-dataset", - "manifest_source_dataset": "institutional/institutional-books-1.0", - "num_train_shards": 24, - "expected_val_shard": "shard_00472.parquet", - "filters": { - "language": "eng", - "min_english_proportion": 0.9, - "year_max_exclusive": 1930, - "reject_invalid_date_types": true, - "ocr_min_inclusive": 90.0, - "ocr_disagreement_max_inclusive": 10.0, - "min_tokenizability": 95.0, - "min_tokens": 500, - "min_chars": 2000, - "min_pages": 3, - "min_sentences": 20, - "undated_rows_rejected": true - } -} \ No newline at end of file diff --git a/experiments/think-d12-r20/tokenizer/token_bytes.pt b/experiments/think-d12-r20/tokenizer/token_bytes.pt deleted file mode 100644 index 01d1ec4aab9e8a7d205c3b3ffbeb8da0e9a62db1..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1 -size 132649 diff --git a/experiments/think-d12-r20/tokenizer/tokenizer.pkl b/experiments/think-d12-r20/tokenizer/tokenizer.pkl deleted file mode 100644 index a17bd392980021628053b95d6425fc556aad527a..0000000000000000000000000000000000000000 --- a/experiments/think-d12-r20/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1 -size 404071 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_000500.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_000500.json deleted file mode 100644 index 76b983909f450a9f8bb62c3217333e4767152458..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,139 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "val_bpb": 1.332522614651975, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "thinkcleaned-d12-1ep-sh26-r11", - "wandb_run_id": "e8375954", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset-cleaned,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints", - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json", - "tokenizer_fingerprint": "0a9922b59b5cb78a", - "git_commit_sha": "7f4c957646ab9d9314d7a7aae4f7efd73aa1ff2e", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "thinkcleaned-d12-1ep-sh26-r11", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-1ep-sh26-r11", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "3d2656cfe49b6048", - "artifact_path": "experiments/thinkcleaned-d12-1ep-sh26-r11" - }, - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "3d2656cfe49b6048" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.332522614651975, - "smooth_train_loss": 3.7214987179114285, - "total_training_time": 1307.1219980716705, - "stage_training_flops": 232547388751872000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 232547388751872000 - } -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_001000.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_001000.json deleted file mode 100644 index 4f044ff47be8f5afc8d65a5eea38203559569714..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,139 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "val_bpb": 1.2404084951472125, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "thinkcleaned-d12-1ep-sh26-r11", - "wandb_run_id": "e8375954", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset-cleaned,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 500, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints", - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json", - "tokenizer_fingerprint": "0a9922b59b5cb78a", - "git_commit_sha": "7f4c957646ab9d9314d7a7aae4f7efd73aa1ff2e", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "thinkcleaned-d12-1ep-sh26-r11", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-1ep-sh26-r11", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "3d2656cfe49b6048", - "artifact_path": "experiments/thinkcleaned-d12-1ep-sh26-r11" - }, - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "3d2656cfe49b6048" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24369538, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24369538 - }, - "loop_state": { - "min_val_bpb": 1.2404084951472125, - "smooth_train_loss": 3.619971321170683, - "total_training_time": 2696.2369639873505, - "stage_training_flops": 465094777503744000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 465094777503744000 - } -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_001500.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_001500.json deleted file mode 100644 index 8246986f3f8fdf924d4705e976382fa5c513442b..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,139 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "val_bpb": 1.185795900077561, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "thinkcleaned-d12-1ep-sh26-r11", - "wandb_run_id": "e8375954", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset-cleaned,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 500, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints", - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json", - "tokenizer_fingerprint": "0a9922b59b5cb78a", - "git_commit_sha": "7f4c957646ab9d9314d7a7aae4f7efd73aa1ff2e", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "thinkcleaned-d12-1ep-sh26-r11", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-1ep-sh26-r11", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "3d2656cfe49b6048", - "artifact_path": "experiments/thinkcleaned-d12-1ep-sh26-r11" - }, - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "3d2656cfe49b6048" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86521538, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86521538 - }, - "loop_state": { - "min_val_bpb": 1.185795900077561, - "smooth_train_loss": 3.3830953942307014, - "total_training_time": 4026.685672521591, - "stage_training_flops": 697642166255616000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 697642166255616000 - } -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_002000.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_002000.json deleted file mode 100644 index 79434ae2af40ba1dd77a5ac64cdf17230d30a5a0..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,139 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "val_bpb": 1.138428837130431, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "thinkcleaned-d12-1ep-sh26-r11", - "wandb_run_id": "e8375954", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset-cleaned,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 500, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints", - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json", - "tokenizer_fingerprint": "0a9922b59b5cb78a", - "git_commit_sha": "7f4c957646ab9d9314d7a7aae4f7efd73aa1ff2e", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "thinkcleaned-d12-1ep-sh26-r11", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-1ep-sh26-r11", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "3d2656cfe49b6048", - "artifact_path": "experiments/thinkcleaned-d12-1ep-sh26-r11" - }, - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "3d2656cfe49b6048" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48673538, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48673538 - }, - "loop_state": { - "min_val_bpb": 1.138428837130431, - "smooth_train_loss": 3.5114756844749833, - "total_training_time": 5358.042316198349, - "stage_training_flops": 930189555007488000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 930189555007488000 - } -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_002362.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_002362.json deleted file mode 100644 index c9051e61a453ef5eb8a7038a04187b32ad2b9355..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,139 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "val_bpb": 1.115457145344029, - "model_config": { - "sequence_len": 2048, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "thinkcleaned-d12-1ep-sh26-r11", - "wandb_run_id": "e8375954", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset-cleaned,d12,ratio11.25", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 2048, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 16, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": 500, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints", - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json", - "tokenizer_fingerprint": "0a9922b59b5cb78a", - "git_commit_sha": "7f4c957646ab9d9314d7a7aae4f7efd73aa1ff2e", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "thinkcleaned-d12-1ep-sh26-r11", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-1ep-sh26-r11", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "3d2656cfe49b6048", - "artifact_path": "experiments/thinkcleaned-d12-1ep-sh26-r11" - }, - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "3d2656cfe49b6048" - }, - "device_batch_size": 16, - "max_seq_len": 2048, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38471586, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38471586 - }, - "loop_state": { - "min_val_bpb": 1.115457145344029, - "smooth_train_loss": 3.2647400994218767, - "total_training_time": 6322.441261768341, - "stage_training_flops": 1098553864463843328, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1098553864463843328 - } -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_000500.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_000500.pt deleted file mode 100644 index 99f85fd3e3bb63d78b85d713ba26da167bdf5de5..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:4b705d047f3a0c23bbc991ace9fb0415b75f81a419e74ad186324eebc49da397 -size 792761690 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_001000.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_001000.pt deleted file mode 100644 index b2d891898605a82b097c7a7f7831a488dad9ae2f..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:df783d2af59fb1d4bf290019d1c9529896b0f9a623d6df9a5859d63535d47b5b -size 792761690 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_001500.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_001500.pt deleted file mode 100644 index 8544c8a7d77099209fb43e62441c1f758ed9bd87..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:837826c2b175cbfc616195d60418500377ffeaf649c2ec5fe3393bdc5bc22ca8 -size 792761690 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_002000.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_002000.pt deleted file mode 100644 index d36f4acefbb6d306f75e582e1b13cdb0544304ad..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f585203c4ba255265381a9e7d757dcbceb30114221054f9cc310334292d652c7 -size 792761690 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_002362.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_002362.pt deleted file mode 100644 index c0d59f7b444c6baf653cbfc6e0279b6897c9da60..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:68d378385029fb2d10e84db5808dd6d9ee8bf7e807b17ab4a47e70e8965542bf -size 792761690 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_000500_rank0.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index 675741712e9c45665e752ee96a1eb2ae5e3b715b..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c3ede0f1c2bdbaf19c446960c445a1c645e35a1aa476749d8d279f3fa825c43e -size 1246165357 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_001000_rank0.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index f3421bd84b57a7548a6bacf45bd8fd26c6e84029..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2607a60fcb952467cd994577a1ecbf2a48701154696d18b24333e9e0a126b866 -size 1246165357 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_001500_rank0.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 21891007cbd9d616966fdbf468b9bedafde5c3cd..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:9872426092acb81cedaccd3527ad5f3b6e864e947bfb0fc13589fea32514658c -size 1246165357 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_002000_rank0.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index 5f17de6bfeab68b303b4e39877e0125a5fdf9bed..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:dea53faac8f5839609bc912f9ce25d64a824d2ea7c5bd922f98ff36c2b8950cf -size 1246165357 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_002362_rank0.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index cfc5566f3ec8219e377453b5802fd3f7d8ddf398..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:e146cae0ec7e4305313eba9e099afc97e789e9c2ba02613ee5a96f769da40a80 -size 1246165357 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json deleted file mode 100644 index f0f840d2413605a8e126c20033b0f778831b4008..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-1ep-sh26-r11", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "3d2656cfe49b6048", - "artifact_path": "experiments/thinkcleaned-d12-1ep-sh26-r11" -} diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb.json deleted file mode 100644 index 068f75caea1d46adc479ace11d57ead0c66c2f8d..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.0608529946072547 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb_on_think-dataset-clean-1930s.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb_on_think-dataset-clean-1930s.json deleted file mode 100644 index be4f020cf240d19e781acc4beeee7d8dd9e28cc2..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb_on_think-dataset-clean-1930s.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.0573472245939946 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb_on_think-dataset.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb_on_think-dataset.json deleted file mode 100644 index c98694dc87f14ca6025b83de7f86f21f7005f7cf..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb_on_think-dataset.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.1045456977914099 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/run.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/run.json deleted file mode 100644 index 73b780dbc0f5ae40b1c2932e3b54ac08a09e8367..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "3d2656cfe49b6048", - "wandb_run_id": "e8375954", - "created_at": 1783267465 -} diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/summary.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/summary.json deleted file mode 100644 index bb6cc23b37f533f308934702fd5dd08ce8456b3c..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/summary.json +++ /dev/null @@ -1,32 +0,0 @@ -{ - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.115457145344029, - "minimum_sampled_val_bpb": 1.115457145344029, - "full_val_bpb": 1.0608529946072547, - "core_metric": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [], - "training_time_seconds": 6322.441261768341, - "stage_training_flops": 1.0985538644638433e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.0985538644638433e+18, - "config_fingerprint": "3d2656cfe49b6048", - "git_commit_sha": "7f4c957646ab9d9314d7a7aae4f7efd73aa1ff2e", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/e8375954", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/thinkcleaned-d12-1ep-sh26-r11", - "dataset_fingerprint": "b839fe2c141e60dd", - "tokenizer_fingerprint": "0a9922b59b5cb78a", - "unique_train_tokens": 1275519304, - "effective_epochs": 0.970873786164196 -} diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/experiment_tokenizer.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 973aef9df6a827762c0983a8a72ca075b0138ee5..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1783267503 -} diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/token_bytes.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/token_bytes.pt deleted file mode 100644 index c9e8a074bb8475e3a950c6a4e610efd20658ef6b..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b24d3f9325f437305a97a56960479be10f222cc70fa50df4fe7f7809b3c51ec8 -size 132649 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/tokenizer.pkl b/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/tokenizer.pkl deleted file mode 100644 index 5975ec52e31c1ab60dd365c96a500f5fb02ae409..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3dc059e668949894537dd6e813704ebb69db930d54de7181d6202f1d60f91636 -size 407574 diff --git a/experiments/thinkcleaned-d12-1ep-sh28-r11/config.json b/experiments/thinkcleaned-d12-1ep-sh28-r11/config.json deleted file mode 100644 index 09926ff3799da9ec6b6551f53d914ade84f4544f..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh28-r11/config.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-1ep-sh28-r11", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-cleaned", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-1ep-sh28-r11", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "6d042a15b4b42e5e", - "artifact_path": "experiments/thinkcleaned-d12-1ep-sh28-r11" -} diff --git a/experiments/thinkcleaned-d12-1ep-sh28-r11/run.json b/experiments/thinkcleaned-d12-1ep-sh28-r11/run.json deleted file mode 100644 index d70e9e04c3dc4d206f3aa976ba058919155ddb74..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh28-r11/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "thinkcleaned-d12-1ep-sh28-r11", - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-1ep-sh28-r11", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "6d042a15b4b42e5e", - "wandb_run_id": "ac903c89", - "created_at": 1783266646 -} diff --git a/experiments/thinkcleaned-d12-1ep-sh29-r11/config.json b/experiments/thinkcleaned-d12-1ep-sh29-r11/config.json deleted file mode 100644 index e586213c3001162e411f71248bfc345fd5f13128..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh29-r11/config.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-1ep-sh29-r11", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "window_pattern": "L", - "device_batch_size": 16, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-1ep-sh29-r11", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25" - ] - }, - "config_fingerprint": "4509e64262826e28", - "artifact_path": "experiments/thinkcleaned-d12-1ep-sh29-r11" -} diff --git a/experiments/thinkcleaned-d12-1ep-sh29-r11/run.json b/experiments/thinkcleaned-d12-1ep-sh29-r11/run.json deleted file mode 100644 index 814f824b061f4dd6510a3a23da50a4cde9b07130..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh29-r11/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "thinkcleaned-d12-1ep-sh29-r11", - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-1ep-sh29-r11", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "4509e64262826e28", - "wandb_run_id": "ba5e5413", - "created_at": 1783266942 -} diff --git a/experiments/thinkcleaned-d12-1ep-sh29-r11/tokenizer/experiment_tokenizer.json b/experiments/thinkcleaned-d12-1ep-sh29-r11/tokenizer/experiment_tokenizer.json deleted file mode 100644 index fb61a5b6c9a2427345ffb5379810da50f3e16c58..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh29-r11/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "thinkcleaned-d12-1ep-sh29-r11", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 24, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1783266959 -} diff --git a/experiments/thinkcleaned-d12-1ep-sh29-r11/tokenizer/token_bytes.pt b/experiments/thinkcleaned-d12-1ep-sh29-r11/tokenizer/token_bytes.pt deleted file mode 100644 index b2f9adce9dedcd5244450bdab92f7277042e8a36..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh29-r11/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:f6bae3dd7ad2f2136963cdfe0e5318c4fb79f4c0e3e2862250bf9a44eb041b48 -size 132649 diff --git a/experiments/thinkcleaned-d12-1ep-sh29-r11/tokenizer/tokenizer.pkl b/experiments/thinkcleaned-d12-1ep-sh29-r11/tokenizer/tokenizer.pkl deleted file mode 100644 index 7f776b7b2a030551507b39a31e007d7d6fbdb46d..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-1ep-sh29-r11/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8cd693c107e8902355b1127575137d5de4c81a275679afc952e2ab7e85f521ab -size 407489 diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_000500.json b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_000500.json deleted file mode 100644 index a4bd222859037ac91dd069d2e5033814f0ff8981..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_000500.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 500, - "training_complete": false, - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "val_bpb": 1.3156279878581487, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "thinkcleaned-d12-r11.25-ctx4096", - "wandb_run_id": "6417ce31", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset-cleaned,d12,ratio11.25,ctx4096", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints", - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/config.json", - "tokenizer_fingerprint": "d2f0e98cb956bfab", - "git_commit_sha": "478579fe478766817e6de504ad0d89bddbbf979a", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "thinkcleaned-d12-r11.25-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-r11.25-ctx4096", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25", - "ctx4096" - ] - }, - "config_fingerprint": "fa1903fa8b505455", - "artifact_path": "experiments/thinkcleaned-d12-r11.25-ctx4096" - }, - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fa1903fa8b505455" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 2, - "pos": 62184769, - "epoch": 1, - "pq_idx": 2, - "rg_idx": 62184769 - }, - "loop_state": { - "min_val_bpb": 1.3156279878581487, - "smooth_train_loss": 3.6770687821378663, - "total_training_time": 1485.2843041419983, - "stage_training_flops": 291921016651776000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 291921016651776000 - } -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_001000.json b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_001000.json deleted file mode 100644 index fd4771c338488e76cb0519b84e2152928fb58d9c..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_001000.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 1000, - "training_complete": false, - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "val_bpb": 1.2163760872799223, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "thinkcleaned-d12-r11.25-ctx4096", - "wandb_run_id": "6417ce31", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset-cleaned,d12,ratio11.25,ctx4096", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints", - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/config.json", - "tokenizer_fingerprint": "d2f0e98cb956bfab", - "git_commit_sha": "478579fe478766817e6de504ad0d89bddbbf979a", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "thinkcleaned-d12-r11.25-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-r11.25-ctx4096", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25", - "ctx4096" - ] - }, - "config_fingerprint": "fa1903fa8b505455", - "artifact_path": "experiments/thinkcleaned-d12-r11.25-ctx4096" - }, - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fa1903fa8b505455" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 5, - "pos": 24336769, - "epoch": 1, - "pq_idx": 5, - "rg_idx": 24336769 - }, - "loop_state": { - "min_val_bpb": 1.2163760872799223, - "smooth_train_loss": 3.5907511967367123, - "total_training_time": 3002.318539381027, - "stage_training_flops": 583842033303552000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 583842033303552000 - } -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_001500.json b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_001500.json deleted file mode 100644 index 64df3905db2e3a2dd3278227ee34bab42c4118bb..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_001500.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 1500, - "training_complete": false, - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "val_bpb": 1.1623499624139166, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "thinkcleaned-d12-r11.25-ctx4096", - "wandb_run_id": "6417ce31", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset-cleaned,d12,ratio11.25,ctx4096", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints", - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/config.json", - "tokenizer_fingerprint": "d2f0e98cb956bfab", - "git_commit_sha": "478579fe478766817e6de504ad0d89bddbbf979a", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "thinkcleaned-d12-r11.25-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-r11.25-ctx4096", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25", - "ctx4096" - ] - }, - "config_fingerprint": "fa1903fa8b505455", - "artifact_path": "experiments/thinkcleaned-d12-r11.25-ctx4096" - }, - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fa1903fa8b505455" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 7, - "pos": 86488769, - "epoch": 1, - "pq_idx": 7, - "rg_idx": 86488769 - }, - "loop_state": { - "min_val_bpb": 1.1623499624139166, - "smooth_train_loss": 3.2949821653997535, - "total_training_time": 4520.007841825485, - "stage_training_flops": 875763049955328000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 875763049955328000 - } -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_002000.json b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_002000.json deleted file mode 100644 index 06d10cd89c615def3797603732313d7e5991df91..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_002000.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 2000, - "training_complete": false, - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "val_bpb": 1.1158280349710101, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "thinkcleaned-d12-r11.25-ctx4096", - "wandb_run_id": "6417ce31", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset-cleaned,d12,ratio11.25,ctx4096", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints", - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/config.json", - "tokenizer_fingerprint": "d2f0e98cb956bfab", - "git_commit_sha": "478579fe478766817e6de504ad0d89bddbbf979a", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "thinkcleaned-d12-r11.25-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-r11.25-ctx4096", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25", - "ctx4096" - ] - }, - "config_fingerprint": "fa1903fa8b505455", - "artifact_path": "experiments/thinkcleaned-d12-r11.25-ctx4096" - }, - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fa1903fa8b505455" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 10, - "pos": 48640769, - "epoch": 1, - "pq_idx": 10, - "rg_idx": 48640769 - }, - "loop_state": { - "min_val_bpb": 1.1158280349710101, - "smooth_train_loss": 3.370424770509588, - "total_training_time": 6038.061463356018, - "stage_training_flops": 1167684066607104000, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1167684066607104000 - } -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_002362.json b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_002362.json deleted file mode 100644 index 97978376d6a9b43e598c1e7004966e9a07a763e7..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/meta_002362.json +++ /dev/null @@ -1,141 +0,0 @@ -{ - "step": 2362, - "training_complete": true, - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "val_bpb": 1.0931769280610826, - "model_config": { - "sequence_len": 4096, - "vocab_size": 32768, - "n_layer": 12, - "n_head": 6, - "n_kv_head": 6, - "n_embd": 768, - "window_pattern": "L" - }, - "user_config": { - "run": "thinkcleaned-d12-r11.25-ctx4096", - "wandb_run_id": "6417ce31", - "wandb_group": "think-d12", - "wandb_tags": "think-dataset-cleaned,d12,ratio11.25,ctx4096", - "device_type": "", - "fp8": false, - "fp8_recipe": "tensorwise", - "depth": 12, - "aspect_ratio": 64, - "head_dim": 128, - "max_seq_len": 4096, - "window_pattern": "L", - "num_iterations": -1, - "target_flops": -1.0, - "target_param_data_ratio": 11.25, - "device_batch_size": 8, - "total_batch_size": 524288, - "embedding_lr": 0.3, - "unembedding_lr": 0.008, - "weight_decay": 0.28, - "matrix_lr": 0.02, - "scalar_lr": 0.5, - "warmup_steps": 40, - "warmdown_ratio": 0.65, - "final_lr_frac": 0.05, - "resume_from_step": -1, - "pretokenized": true, - "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/data", - "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer", - "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/pretok", - "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints", - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-r11.25-ctx4096/config.json", - "tokenizer_fingerprint": "d2f0e98cb956bfab", - "git_commit_sha": "478579fe478766817e6de504ad0d89bddbbf979a", - "seed": 42, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "core_metric_max_per_task": 500, - "sample_every": -1, - "save_every": 500, - "model_tag": "thinkcleaned-d12-r11.25-ctx4096", - "experiment": { - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-r11.25-ctx4096", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25", - "ctx4096" - ] - }, - "config_fingerprint": "fa1903fa8b505455", - "artifact_path": "experiments/thinkcleaned-d12-r11.25-ctx4096" - }, - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fa1903fa8b505455" - }, - "device_batch_size": 8, - "max_seq_len": 4096, - "total_batch_size": 524288, - "dataloader_state_dict": { - "file_idx": 12, - "pos": 38438817, - "epoch": 1, - "pq_idx": 12, - "rg_idx": 38438817 - }, - "loop_state": { - "min_val_bpb": 1.0931769280610826, - "smooth_train_loss": 3.163295955295285, - "total_training_time": 7138.363130569458, - "stage_training_flops": 1379034882662989824, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1379034882662989824 - } -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_000500.pt b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_000500.pt deleted file mode 100644 index 6833ab1b89e1cfc681ddc1616a010bf270950c39..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_000500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:64488e1d1324f6b3eee691daa3ae419b9537433aaaaacfad927ba0d999e4414f -size 792761690 diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_001000.pt b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_001000.pt deleted file mode 100644 index c558537fdb540125b35764b6d9836897963b868b..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_001000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:d8b02851e6bf9c45a2c82e45b69170e0599b79eff82aa0d6bc611f86e0df1d49 -size 792761690 diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_001500.pt b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_001500.pt deleted file mode 100644 index e351960f1f6b3446b8624df42a48991ed50e0426..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_001500.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a3f94529c91b3ec91d49a1db14a3414955b1194b30d63a7e67a08d728de9acae -size 792761690 diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_002000.pt b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_002000.pt deleted file mode 100644 index db4e37ae45dffca34aecc3e35b1994ba61188769..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_002000.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:67ac4704cad1e4415ddf0fb0049d2c1cb3705e45f26f2c8631534c8c6fcb8c2d -size 792761690 diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_002362.pt b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_002362.pt deleted file mode 100644 index 4452240c9d66dd288ad50583c6e1ce38f2f89abe..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/model_002362.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:c46264d7b3baf9dd925b2d3ebd9f017ed7df60d8e6115bc67fdf4f450dcf988d -size 792761690 diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_000500_rank0.pt b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_000500_rank0.pt deleted file mode 100644 index a2a5f1f8b5da2d0defca12ca6816484ee5e7394b..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_000500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:cf1521ebb71bcb7d2061a9d6a2839ab11516f5262d38fb71903432ff4c64f3a1 -size 1246165357 diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_001000_rank0.pt b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_001000_rank0.pt deleted file mode 100644 index 70bcd9dc8de4eae7d3423cd24fcbd7dbf64c7e88..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_001000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:95895aabfaec8492c8a492e0a8dd8f9fd4e9cbfaff93a523fdea52c6ee378b54 -size 1246165357 diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_001500_rank0.pt b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_001500_rank0.pt deleted file mode 100644 index 7c162845a14f2cf9f23c3bec6a848de8525be33f..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_001500_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:477579d7aa0201b5d468232356730898c9e0dc6d1c2d79906ee190a826f8b4e5 -size 1246165357 diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_002000_rank0.pt b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_002000_rank0.pt deleted file mode 100644 index f00a7ff6b5da7bbe82fa7a2d98c85f4972760cc1..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_002000_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8464dcf968c888f6b6d58cc82a54142e2643a69aad42387acedcd4c1e1f1b7bb -size 1246165357 diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_002362_rank0.pt b/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_002362_rank0.pt deleted file mode 100644 index 4a58b167167ddb70f6ac6453dcbe4c0590bcf74b..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/base_checkpoints/optim_002362_rank0.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8e5d09190ec8e7d167716699beeeeaaa6a18638183bd7f6520900fdea9e668cb -size 1246165357 diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/config.json b/experiments/thinkcleaned-d12-r11.25-ctx4096/config.json deleted file mode 100644 index 02884732c358524b93ce5f33bc281340f9e0dc26..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/config.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "schema_version": 1, - "stage": "base", - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "pretokenize": { - "enabled": true, - "slack": 1.03, - "val_tokens": 20971520, - "shard_tokens": 100000000, - "tokenizer_threads": 8 - }, - "training": { - "depth": 12, - "scaling_params": 110100912, - "target_param_data_ratio": 11.25, - "max_seq_len": 4096, - "window_pattern": "L", - "device_batch_size": 8, - "total_batch_size": 524288, - "seed": 42, - "save_every": 500, - "eval_every": 250, - "eval_tokens": 2097152, - "core_metric_every": -1, - "sample_every": -1 - }, - "artifacts": { - "repo": "jbduran/think.nano" - }, - "wandb": { - "entity": "jbduran-thinkingmachinesncsu", - "project": "think.nano", - "name": "thinkcleaned-d12-r11.25-ctx4096", - "group": "think-d12", - "tags": [ - "think-dataset-cleaned", - "d12", - "ratio11.25", - "ctx4096" - ] - }, - "config_fingerprint": "fa1903fa8b505455", - "artifact_path": "experiments/thinkcleaned-d12-r11.25-ctx4096" -} diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/evals/core.json b/experiments/thinkcleaned-d12-r11.25-ctx4096/evals/core.json deleted file mode 100644 index 1a629cce9cd6adeffc69308dcc943756b501d033..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/evals/core.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": 0.0721177661677981, - "core_results": { - "hellaswag_zeroshot": 0.27663812041282654, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.06968160718679428, - "arc_easy": 0.3249158263206482, - "arc_challenge": 0.22184298932552338, - "copa": 0.5299999713897705, - "commonsense_qa": 0.2858312726020813, - "piqa": 0.5473340153694153, - "openbook_qa": 0.2540000081062317, - "lambada_openai": 0.20648165047168732, - "hellaswag": 0.27912765741348267, - "winograd": 0.5714285969734192, - "winogrande": 0.5114443302154541, - "bigbench_dyck_languages": 0.09100000560283661, - "agi_eval_lsat_ar": 0.260869562625885, - "bigbench_cs_algorithms": 0.40984848141670227, - "bigbench_operators": 0.0476190485060215, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.010879848152399063, - "coqa": 0.07052486389875412, - "boolq": 0.5629969239234924, - "bigbench_language_identification": 0.25849997997283936 - }, - "centered_results": { - "hellaswag_zeroshot": 0.035517493883768715, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.06968160718679428, - "arc_easy": 0.09988776842753093, - "arc_challenge": -0.03754268089930216, - "copa": 0.059999942779541016, - "commonsense_qa": 0.10728909075260161, - "piqa": 0.09466803073883057, - "openbook_qa": 0.005333344141642253, - "lambada_openai": 0.20648165047168732, - "hellaswag": 0.03883687655131022, - "winograd": 0.14285719394683838, - "winogrande": 0.022888660430908203, - "bigbench_dyck_languages": 0.09100000560283661, - "agi_eval_lsat_ar": 0.07608695328235625, - "bigbench_cs_algorithms": 0.40984848141670227, - "bigbench_operators": 0.0476190485060215, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.010879848152399063, - "coqa": 0.07052486389875412, - "boolq": -0.1500080949381778, - "bigbench_language_identification": 0.18426840481060436 - }, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/evals/samples.json b/experiments/thinkcleaned-d12-r11.25-ctx4096/evals/samples.json deleted file mode 100644 index 15c38be6a65e6cbc72fa3ce4025e7d4ce7484ff2..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/evals/samples.json +++ /dev/null @@ -1,48 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": {}, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the United States, and the capital of the United States is the" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as that of silver, and the same as that of gold. The" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Friday. The\n\nLord's day is the day of the Lord's\n\nSu" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as hot, and the same as hot. The same word is used" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the sun, the moon, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the same as the \"sweetness of the air,\" and the \"s" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the feet of the body, and the number of feet of the" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nto the works of an author, who has made us the followers of another and of a different school. It had great influence in the University and the Church; but in the prime of education there was hardly a University teacher-at least no I. KIL.BC was taught in any University.\n\nThe Principles of our Composition.-I desire, therefore, to say the\n\nDedication of the Elements and Elements of a Reading.\n\nI I\n\nTHE FIRST SECTION OF THE GENERAL treatise of the Epistle was written in a peculiar manner, which required only a Bibliography from the author of it to be complete.\n\nThis also bore upon habits of thought and", - "<|bos|>As iron steamer, double iron Cheston, one of the items of powder in the North Star, to be the first imparted to town, passed his.\n\nHowse rushing with impetuosity as he proceeded! At all points magnetic influence and agency, exercises for a moment; but oh! then afterwards, by mere force of force man becomes not only moral but magnetic.\n\nWolverhampton could have easily enough rendered the country very safe from attack, but on his arrival at Harrisburg, he found the genuine Roman magnet not perfectly made out, and apparently launches along the extremities of the ever-smoothed Western Atlantic.\n\nBok", - "<|bos|>ady Christians, Sanghahrochs with one or two of her youths and who are the most affectionate, seem to start these fine ladies with a very languid and all eloquent manner, and excite the imaginations of the people, who are somewhat startled by it. No one could be more realistic than Tunghah.\n\nAnd when Miss Jermyn and her daughters heard-or in the light of another spirit felt then the power of such eloquence-they sent a sacred present to the holy being to \"translate\" him. That is not seen in this world, and can find no place in it on earth.\n\nValiantly ladies", - "<|bos|> fanaticism and fanaticism which has given such deep and varied expression among Christians; and it shows itself in its efforts to produce a future as distinguished from present amelioration as we contemplate this question of crime and punishment, the instrumentality employed in that elevated system of morality which is embodied in their laws and established laws. By considering\n\nit as finally fettered in the dominion of an immaterial, eternal reason, touching it as among the mysteries of the universe, by which mortal men in all civilized countries are regulated and admonished, we are prepared to verify assertors of this vital doctrine a number of false and false facts. I will now only add", - "<|bos|>PREFACE.\n\nCONTENTS.\n\nvii ...415 599 610 462 151-161 Charles II., King of Spain, and prosecutor of the French, ca. 143-168.\n\n1741794, king of Poland, excommunicated and imprisoned in France and Germany, 18-19\n\n178-190 Charles I., King of France, and His ministers, Gen. John Gen1, 190-191 Charles II., King of France, and His French ministers, Count de M\u00e9di\u00e9re, 192-193 George III.,", - "<|bos|>PREFACE.\n\nIf any of these places were provided the expedition was understood to have had for its object a simple expedition into the interior, for the capture of which the war is responsible.\n\nThis war without doubt won much greater and purer victories to our country than some of those campaigns experienced in the last two years. The present war is an epoch in the history of this country because in it the great difficulty of driving superior fleets and troops overcomes the ignorance and prejudices of those who have to enter it in time of political transition; it is no war because in overcoming difficulties we have immediately to find and seek (a thing in the simple military", - "<|bos|>ated have the consideration more than the influence of fortification, which is always the wisest way of awakening and gratifying appetite, and which, if the means be 1\n\ncieved, ameliorates and simplifies the symptoms of a prescribed form of life that it has no warlike sound.\n\nCheops was composed by Sir H. Walpole of Brown's bones; three are believed to be the four grand points in the history of the human race that will never live in aeons, and send certain death to their kind.\n\nThere remains the vascular), while its beauties were so congenial to the southern society to which they belong, that the", - "<|bos|>PREFACE.\n\nMODERN REPRODUCTION.\n\nThe author has a strong and conscientious predilection for the theory of crude reproduction. He therefore presents, with some of his more popular works, a collection of facts on this branch of moral science. The collected editions of his works have been published privately. Hamilton has laboured during the past sixteen years to establish a system of indirectly controllable morals, which embodies at best a partial abridgment of the first six volumes of moral Biology; and Consent to the institution of Plurality of Worldly Honor, having a strong personal interest in the present agitation for the existing creation of our species, and such an interest as" - ] -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/evals/val_bpb.json b/experiments/thinkcleaned-d12-r11.25-ctx4096/evals/val_bpb.json deleted file mode 100644 index 231d3a3b7b9fed5efc9aff7f145c2663743b6961..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/evals/val_bpb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "model": "base_model (step 2362)", - "step": 2362, - "bpb": { - "val": 1.0433092019869723 - }, - "core_metric": null, - "core_results": null, - "centered_results": null, - "conditioned_samples": [], - "unconditioned_samples": [] -} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/run.json b/experiments/thinkcleaned-d12-r11.25-ctx4096/run.json deleted file mode 100644 index 7b1429a7ebe69bc4761150847634ce5b49e17f6e..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/run.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "config_fingerprint": "fa1903fa8b505455", - "wandb_run_id": "6417ce31", - "created_at": 1783693513 -} diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/summary.json b/experiments/thinkcleaned-d12-r11.25-ctx4096/summary.json deleted file mode 100644 index cdaf6c009114b1b34dfd7d8d7e1df8a9fd4e9280..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/summary.json +++ /dev/null @@ -1,93 +0,0 @@ -{ - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "stage": "base", - "base_experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "parent_experiment_id": null, - "parent_checkpoint_step": null, - "dataset": "jbduran/think-dataset-clean", - "dataset_revision": "main", - "step": 2362, - "depth": 12, - "target_param_data_ratio": 11.25, - "training_tokens": 1238368256, - "final_sampled_val_bpb": 1.0931769280610826, - "minimum_sampled_val_bpb": 1.0931769280610826, - "full_val_bpb": 1.0433092019869723, - "core_metric": 0.0721177661677981, - "centered_results": { - "hellaswag_zeroshot": 0.035517493883768715, - "jeopardy": 0.0004723665479104966, - "bigbench_qa_wikidata": 0.06968160718679428, - "arc_easy": 0.09988776842753093, - "arc_challenge": -0.03754268089930216, - "copa": 0.059999942779541016, - "commonsense_qa": 0.10728909075260161, - "piqa": 0.09466803073883057, - "openbook_qa": 0.005333344141642253, - "lambada_openai": 0.20648165047168732, - "hellaswag": 0.03883687655131022, - "winograd": 0.14285719394683838, - "winogrande": 0.022888660430908203, - "bigbench_dyck_languages": 0.09100000560283661, - "agi_eval_lsat_ar": 0.07608695328235625, - "bigbench_cs_algorithms": 0.40984848141670227, - "bigbench_operators": 0.0476190485060215, - "bigbench_repeat_copy_logic": 0.0, - "squad": 0.010879848152399063, - "coqa": 0.07052486389875412, - "boolq": -0.1500080949381778, - "bigbench_language_identification": 0.18426840481060436 - }, - "conditioned_samples": [ - { - "prompt": "The capital of France is", - "text": "<|bos|>The capital of France is the capital of the United States, and the capital of the United States is the" - }, - { - "prompt": "The chemical symbol of gold is", - "text": "<|bos|>The chemical symbol of gold is the same as that of silver, and the same as that of gold. The" - }, - { - "prompt": "If yesterday was Friday, then tomorrow will be", - "text": "<|bos|>If yesterday was Friday, then tomorrow will be Friday. The\n\nLord's day is the day of the Lord's\n\nSu" - }, - { - "prompt": "The opposite of hot is", - "text": "<|bos|>The opposite of hot is the same as hot, and the same as hot. The same word is used" - }, - { - "prompt": "The planets of the solar system are:", - "text": "<|bos|>The planets of the solar system are: the sun, the moon, the stars, the sun, the moon, the" - }, - { - "prompt": "My favorite color is", - "text": "<|bos|>My favorite color is the same as the \"sweetness of the air,\" and the \"s" - }, - { - "prompt": "If 5*x + 3 = 13, then x is", - "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the feet of the body, and the number of feet of the" - } - ], - "unconditioned_samples": [ - "<|bos|>PREFACE.\n\nto the works of an author, who has made us the followers of another and of a different school. It had great influence in the University and the Church; but in the prime of education there was hardly a University teacher-at least no I. KIL.BC was taught in any University.\n\nThe Principles of our Composition.-I desire, therefore, to say the\n\nDedication of the Elements and Elements of a Reading.\n\nI I\n\nTHE FIRST SECTION OF THE GENERAL treatise of the Epistle was written in a peculiar manner, which required only a Bibliography from the author of it to be complete.\n\nThis also bore upon habits of thought and", - "<|bos|>As iron steamer, double iron Cheston, one of the items of powder in the North Star, to be the first imparted to town, passed his.\n\nHowse rushing with impetuosity as he proceeded! At all points magnetic influence and agency, exercises for a moment; but oh! then afterwards, by mere force of force man becomes not only moral but magnetic.\n\nWolverhampton could have easily enough rendered the country very safe from attack, but on his arrival at Harrisburg, he found the genuine Roman magnet not perfectly made out, and apparently launches along the extremities of the ever-smoothed Western Atlantic.\n\nBok", - "<|bos|>ady Christians, Sanghahrochs with one or two of her youths and who are the most affectionate, seem to start these fine ladies with a very languid and all eloquent manner, and excite the imaginations of the people, who are somewhat startled by it. No one could be more realistic than Tunghah.\n\nAnd when Miss Jermyn and her daughters heard-or in the light of another spirit felt then the power of such eloquence-they sent a sacred present to the holy being to \"translate\" him. That is not seen in this world, and can find no place in it on earth.\n\nValiantly ladies", - "<|bos|> fanaticism and fanaticism which has given such deep and varied expression among Christians; and it shows itself in its efforts to produce a future as distinguished from present amelioration as we contemplate this question of crime and punishment, the instrumentality employed in that elevated system of morality which is embodied in their laws and established laws. By considering\n\nit as finally fettered in the dominion of an immaterial, eternal reason, touching it as among the mysteries of the universe, by which mortal men in all civilized countries are regulated and admonished, we are prepared to verify assertors of this vital doctrine a number of false and false facts. I will now only add", - "<|bos|>PREFACE.\n\nCONTENTS.\n\nvii ...415 599 610 462 151-161 Charles II., King of Spain, and prosecutor of the French, ca. 143-168.\n\n1741794, king of Poland, excommunicated and imprisoned in France and Germany, 18-19\n\n178-190 Charles I., King of France, and His ministers, Gen. John Gen1, 190-191 Charles II., King of France, and His French ministers, Count de M\u00e9di\u00e9re, 192-193 George III.,", - "<|bos|>PREFACE.\n\nIf any of these places were provided the expedition was understood to have had for its object a simple expedition into the interior, for the capture of which the war is responsible.\n\nThis war without doubt won much greater and purer victories to our country than some of those campaigns experienced in the last two years. The present war is an epoch in the history of this country because in it the great difficulty of driving superior fleets and troops overcomes the ignorance and prejudices of those who have to enter it in time of political transition; it is no war because in overcoming difficulties we have immediately to find and seek (a thing in the simple military", - "<|bos|>ated have the consideration more than the influence of fortification, which is always the wisest way of awakening and gratifying appetite, and which, if the means be 1\n\ncieved, ameliorates and simplifies the symptoms of a prescribed form of life that it has no warlike sound.\n\nCheops was composed by Sir H. Walpole of Brown's bones; three are believed to be the four grand points in the history of the human race that will never live in aeons, and send certain death to their kind.\n\nThere remains the vascular), while its beauties were so congenial to the southern society to which they belong, that the", - "<|bos|>PREFACE.\n\nMODERN REPRODUCTION.\n\nThe author has a strong and conscientious predilection for the theory of crude reproduction. He therefore presents, with some of his more popular works, a collection of facts on this branch of moral science. The collected editions of his works have been published privately. Hamilton has laboured during the past sixteen years to establish a system of indirectly controllable morals, which embodies at best a partial abridgment of the first six volumes of moral Biology; and Consent to the institution of Plurality of Worldly Honor, having a strong personal interest in the present agitation for the existing creation of our species, and such an interest as" - ], - "training_time_seconds": 7138.363130569458, - "stage_training_flops": 1.3790348826629898e+18, - "inherited_parent_flops": 0.0, - "cumulative_pipeline_training_flops": 1.3790348826629898e+18, - "config_fingerprint": "fa1903fa8b505455", - "git_commit_sha": "478579fe478766817e6de504ad0d89bddbbf979a", - "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/6417ce31", - "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/thinkcleaned-d12-r11.25-ctx4096", - "dataset_fingerprint": "b839fe2c141e60dd", - "tokenizer_fingerprint": "d2f0e98cb956bfab", - "unique_train_tokens": 1275519304, - "effective_epochs": 0.970873786164196 -} diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer/experiment_tokenizer.json b/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer/experiment_tokenizer.json deleted file mode 100644 index 9ab859e3e3663e4f11bf7fdc6527900784f50a8c..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer/experiment_tokenizer.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "experiment_id": "thinkcleaned-d12-r11.25-ctx4096", - "dataset": { - "adapter": "parquet_shards", - "repo": "jbduran/think-dataset-clean", - "revision": "main", - "validation_shard": 472, - "num_train_shards": 26, - "download_workers": 4 - }, - "tokenizer": { - "mode": "train", - "max_chars": 2000000000, - "doc_cap": 10000, - "vocab_size": 32768 - }, - "created_at": 1783693572 -} diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer/token_bytes.pt b/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer/token_bytes.pt deleted file mode 100644 index c9e8a074bb8475e3a950c6a4e610efd20658ef6b..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer/token_bytes.pt +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b24d3f9325f437305a97a56960479be10f222cc70fa50df4fe7f7809b3c51ec8 -size 132649 diff --git a/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl b/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl deleted file mode 100644 index 5975ec52e31c1ab60dd365c96a500f5fb02ae409..0000000000000000000000000000000000000000 --- a/experiments/thinkcleaned-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:3dc059e668949894537dd6e813704ebb69db930d54de7181d6202f1d60f91636 -size 407574