Import experiment archive from bart (batch 6)
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- experiments/think-d12-r11.25-ctx4096/tokenizer/token_bytes.pt +3 -0
- experiments/think-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl +3 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_000500.json +140 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001000.json +140 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001500.json +140 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002000.json +140 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002362.json +140 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_000500.pt +3 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001000.pt +3 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001500.pt +3 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002000.pt +3 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002362.pt +3 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_000500_rank0.pt +3 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001000_rank0.pt +3 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001500_rank0.pt +3 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002000_rank0.pt +3 -0
- experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002362_rank0.pt +3 -0
- experiments/think-d12-r11.25-ctx8192/config.json +57 -0
- experiments/think-d12-r11.25-ctx8192/evals/core.json +56 -0
- experiments/think-d12-r11.25-ctx8192/evals/samples.json +48 -0
- experiments/think-d12-r11.25-ctx8192/evals/val_bpb.json +174 -0
- experiments/think-d12-r11.25-ctx8192/run.json +10 -0
- experiments/think-d12-r11.25-ctx8192/summary.json +69 -0
- experiments/think-d12-r11.25-ctx8192/tokenizer/experiment_tokenizer.json +18 -0
- experiments/think-d12-r11.25-ctx8192/tokenizer/token_bytes.pt +3 -0
- experiments/think-d12-r11.25-ctx8192/tokenizer/tokenizer.pkl +3 -0
- experiments/think-d12-r11.25/base_checkpoints/meta_000500.json +61 -0
- experiments/think-d12-r11.25/base_checkpoints/meta_001000.json +61 -0
- experiments/think-d12-r11.25/base_checkpoints/meta_001500.json +61 -0
- experiments/think-d12-r11.25/base_checkpoints/meta_002000.json +61 -0
- experiments/think-d12-r11.25/base_checkpoints/meta_002362.json +61 -0
- experiments/think-d12-r11.25/base_checkpoints/model_000500.pt +3 -0
- experiments/think-d12-r11.25/base_checkpoints/model_001000.pt +3 -0
- experiments/think-d12-r11.25/base_checkpoints/model_001500.pt +3 -0
- experiments/think-d12-r11.25/base_checkpoints/model_002000.pt +3 -0
- experiments/think-d12-r11.25/base_checkpoints/model_002362.pt +3 -0
- experiments/think-d12-r11.25/base_checkpoints/optim_000500_rank0.pt +3 -0
- experiments/think-d12-r11.25/base_checkpoints/optim_001000_rank0.pt +3 -0
- experiments/think-d12-r11.25/base_checkpoints/optim_001500_rank0.pt +3 -0
- experiments/think-d12-r11.25/base_checkpoints/optim_002000_rank0.pt +3 -0
- experiments/think-d12-r11.25/base_checkpoints/optim_002362_rank0.pt +3 -0
- experiments/think-d12-r11.25/config.json +55 -0
- experiments/think-d12-r11.25/evals/samples.json +48 -0
- experiments/think-d12-r11.25/evals/val_bpb.json +54 -0
- experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean-1930s.json +12 -0
- experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json +12 -0
- experiments/think-d12-r11.25/run.json +6 -0
- experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/meta_001065.json +38 -0
- experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/model_001065.pt +3 -0
- experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/optim_001065_rank0.pt +3 -0
experiments/think-d12-r11.25-ctx4096/tokenizer/token_bytes.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1
|
| 3 |
+
size 132649
|
experiments/think-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1
|
| 3 |
+
size 404071
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_000500.json
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 500,
|
| 3 |
+
"training_complete": false,
|
| 4 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 5 |
+
"val_bpb": 1.2806006574422366,
|
| 6 |
+
"model_config": {
|
| 7 |
+
"sequence_len": 8192,
|
| 8 |
+
"vocab_size": 32768,
|
| 9 |
+
"n_layer": 12,
|
| 10 |
+
"n_head": 6,
|
| 11 |
+
"n_kv_head": 6,
|
| 12 |
+
"n_embd": 768,
|
| 13 |
+
"window_pattern": "L"
|
| 14 |
+
},
|
| 15 |
+
"user_config": {
|
| 16 |
+
"run": "think-d12-r11.25-ctx8192",
|
| 17 |
+
"wandb_run_id": "e3483a4b",
|
| 18 |
+
"wandb_group": "think-d12",
|
| 19 |
+
"wandb_tags": "think-dataset,d12,ratio11.25,ctx8192",
|
| 20 |
+
"device_type": "",
|
| 21 |
+
"fp8": false,
|
| 22 |
+
"fp8_recipe": "tensorwise",
|
| 23 |
+
"depth": 12,
|
| 24 |
+
"aspect_ratio": 64,
|
| 25 |
+
"head_dim": 128,
|
| 26 |
+
"max_seq_len": 8192,
|
| 27 |
+
"window_pattern": "L",
|
| 28 |
+
"num_iterations": -1,
|
| 29 |
+
"target_flops": -1.0,
|
| 30 |
+
"target_param_data_ratio": 11.25,
|
| 31 |
+
"device_batch_size": 4,
|
| 32 |
+
"total_batch_size": 524288,
|
| 33 |
+
"embedding_lr": 0.3,
|
| 34 |
+
"unembedding_lr": 0.008,
|
| 35 |
+
"weight_decay": 0.28,
|
| 36 |
+
"matrix_lr": 0.02,
|
| 37 |
+
"scalar_lr": 0.5,
|
| 38 |
+
"warmup_steps": 40,
|
| 39 |
+
"warmdown_ratio": 0.65,
|
| 40 |
+
"final_lr_frac": 0.05,
|
| 41 |
+
"resume_from_step": -1,
|
| 42 |
+
"pretokenized": true,
|
| 43 |
+
"data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data",
|
| 44 |
+
"tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer",
|
| 45 |
+
"pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok",
|
| 46 |
+
"checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints",
|
| 47 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 48 |
+
"experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json",
|
| 49 |
+
"tokenizer_fingerprint": "03c4f62e7a9d0c3b",
|
| 50 |
+
"git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847",
|
| 51 |
+
"seed": 42,
|
| 52 |
+
"eval_every": 250,
|
| 53 |
+
"eval_tokens": 2097152,
|
| 54 |
+
"core_metric_every": -1,
|
| 55 |
+
"core_metric_max_per_task": 500,
|
| 56 |
+
"sample_every": -1,
|
| 57 |
+
"save_every": 500,
|
| 58 |
+
"model_tag": "think-d12-r11.25-ctx8192",
|
| 59 |
+
"experiment": {
|
| 60 |
+
"schema_version": 1,
|
| 61 |
+
"stage": "base",
|
| 62 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 63 |
+
"dataset": {
|
| 64 |
+
"adapter": "parquet_shards",
|
| 65 |
+
"repo": "jbduran/think-dataset",
|
| 66 |
+
"revision": "main",
|
| 67 |
+
"validation_shard": 472,
|
| 68 |
+
"num_train_shards": 24,
|
| 69 |
+
"download_workers": 4
|
| 70 |
+
},
|
| 71 |
+
"tokenizer": {
|
| 72 |
+
"mode": "train",
|
| 73 |
+
"max_chars": 2000000000,
|
| 74 |
+
"doc_cap": 10000,
|
| 75 |
+
"vocab_size": 32768
|
| 76 |
+
},
|
| 77 |
+
"pretokenize": {
|
| 78 |
+
"enabled": true,
|
| 79 |
+
"slack": 1.03,
|
| 80 |
+
"val_tokens": 20971520,
|
| 81 |
+
"shard_tokens": 100000000,
|
| 82 |
+
"tokenizer_threads": 8
|
| 83 |
+
},
|
| 84 |
+
"training": {
|
| 85 |
+
"depth": 12,
|
| 86 |
+
"scaling_params": 110100912,
|
| 87 |
+
"target_param_data_ratio": 11.25,
|
| 88 |
+
"max_seq_len": 8192,
|
| 89 |
+
"window_pattern": "L",
|
| 90 |
+
"device_batch_size": 4,
|
| 91 |
+
"total_batch_size": 524288,
|
| 92 |
+
"save_every": 500,
|
| 93 |
+
"eval_every": 250,
|
| 94 |
+
"eval_tokens": 2097152,
|
| 95 |
+
"core_metric_every": -1,
|
| 96 |
+
"sample_every": -1
|
| 97 |
+
},
|
| 98 |
+
"artifacts": {
|
| 99 |
+
"repo": "jbduran/think.nano"
|
| 100 |
+
},
|
| 101 |
+
"wandb": {
|
| 102 |
+
"entity": "jbduran-thinkingmachinesncsu",
|
| 103 |
+
"project": "think.nano",
|
| 104 |
+
"name": "think-d12-r11.25-ctx8192",
|
| 105 |
+
"group": "think-d12",
|
| 106 |
+
"tags": [
|
| 107 |
+
"think-dataset",
|
| 108 |
+
"d12",
|
| 109 |
+
"ratio11.25",
|
| 110 |
+
"ctx8192"
|
| 111 |
+
]
|
| 112 |
+
},
|
| 113 |
+
"config_fingerprint": "407a5074e0bf3730",
|
| 114 |
+
"artifact_path": "experiments/think-d12-r11.25-ctx8192"
|
| 115 |
+
},
|
| 116 |
+
"stage": "base",
|
| 117 |
+
"base_experiment_id": "think-d12-r11.25-ctx8192",
|
| 118 |
+
"parent_experiment_id": null,
|
| 119 |
+
"parent_checkpoint_step": null,
|
| 120 |
+
"config_fingerprint": "407a5074e0bf3730"
|
| 121 |
+
},
|
| 122 |
+
"device_batch_size": 4,
|
| 123 |
+
"max_seq_len": 8192,
|
| 124 |
+
"total_batch_size": 524288,
|
| 125 |
+
"dataloader_state_dict": {
|
| 126 |
+
"file_idx": 2,
|
| 127 |
+
"pos": 62184769,
|
| 128 |
+
"epoch": 1,
|
| 129 |
+
"pq_idx": 2,
|
| 130 |
+
"rg_idx": 62184769
|
| 131 |
+
},
|
| 132 |
+
"loop_state": {
|
| 133 |
+
"min_val_bpb": 1.2806006574422366,
|
| 134 |
+
"smooth_train_loss": 3.5198863114259193,
|
| 135 |
+
"total_training_time": 1859.6971344947815,
|
| 136 |
+
"stage_training_flops": 410668272451584000,
|
| 137 |
+
"inherited_parent_flops": 0.0,
|
| 138 |
+
"cumulative_pipeline_training_flops": 410668272451584000
|
| 139 |
+
}
|
| 140 |
+
}
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001000.json
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 1000,
|
| 3 |
+
"training_complete": false,
|
| 4 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 5 |
+
"val_bpb": 1.1673074973592756,
|
| 6 |
+
"model_config": {
|
| 7 |
+
"sequence_len": 8192,
|
| 8 |
+
"vocab_size": 32768,
|
| 9 |
+
"n_layer": 12,
|
| 10 |
+
"n_head": 6,
|
| 11 |
+
"n_kv_head": 6,
|
| 12 |
+
"n_embd": 768,
|
| 13 |
+
"window_pattern": "L"
|
| 14 |
+
},
|
| 15 |
+
"user_config": {
|
| 16 |
+
"run": "think-d12-r11.25-ctx8192",
|
| 17 |
+
"wandb_run_id": "e3483a4b",
|
| 18 |
+
"wandb_group": "think-d12",
|
| 19 |
+
"wandb_tags": "think-dataset,d12,ratio11.25,ctx8192",
|
| 20 |
+
"device_type": "",
|
| 21 |
+
"fp8": false,
|
| 22 |
+
"fp8_recipe": "tensorwise",
|
| 23 |
+
"depth": 12,
|
| 24 |
+
"aspect_ratio": 64,
|
| 25 |
+
"head_dim": 128,
|
| 26 |
+
"max_seq_len": 8192,
|
| 27 |
+
"window_pattern": "L",
|
| 28 |
+
"num_iterations": -1,
|
| 29 |
+
"target_flops": -1.0,
|
| 30 |
+
"target_param_data_ratio": 11.25,
|
| 31 |
+
"device_batch_size": 4,
|
| 32 |
+
"total_batch_size": 524288,
|
| 33 |
+
"embedding_lr": 0.3,
|
| 34 |
+
"unembedding_lr": 0.008,
|
| 35 |
+
"weight_decay": 0.28,
|
| 36 |
+
"matrix_lr": 0.02,
|
| 37 |
+
"scalar_lr": 0.5,
|
| 38 |
+
"warmup_steps": 40,
|
| 39 |
+
"warmdown_ratio": 0.65,
|
| 40 |
+
"final_lr_frac": 0.05,
|
| 41 |
+
"resume_from_step": -1,
|
| 42 |
+
"pretokenized": true,
|
| 43 |
+
"data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data",
|
| 44 |
+
"tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer",
|
| 45 |
+
"pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok",
|
| 46 |
+
"checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints",
|
| 47 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 48 |
+
"experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json",
|
| 49 |
+
"tokenizer_fingerprint": "03c4f62e7a9d0c3b",
|
| 50 |
+
"git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847",
|
| 51 |
+
"seed": 42,
|
| 52 |
+
"eval_every": 250,
|
| 53 |
+
"eval_tokens": 2097152,
|
| 54 |
+
"core_metric_every": -1,
|
| 55 |
+
"core_metric_max_per_task": 500,
|
| 56 |
+
"sample_every": -1,
|
| 57 |
+
"save_every": 500,
|
| 58 |
+
"model_tag": "think-d12-r11.25-ctx8192",
|
| 59 |
+
"experiment": {
|
| 60 |
+
"schema_version": 1,
|
| 61 |
+
"stage": "base",
|
| 62 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 63 |
+
"dataset": {
|
| 64 |
+
"adapter": "parquet_shards",
|
| 65 |
+
"repo": "jbduran/think-dataset",
|
| 66 |
+
"revision": "main",
|
| 67 |
+
"validation_shard": 472,
|
| 68 |
+
"num_train_shards": 24,
|
| 69 |
+
"download_workers": 4
|
| 70 |
+
},
|
| 71 |
+
"tokenizer": {
|
| 72 |
+
"mode": "train",
|
| 73 |
+
"max_chars": 2000000000,
|
| 74 |
+
"doc_cap": 10000,
|
| 75 |
+
"vocab_size": 32768
|
| 76 |
+
},
|
| 77 |
+
"pretokenize": {
|
| 78 |
+
"enabled": true,
|
| 79 |
+
"slack": 1.03,
|
| 80 |
+
"val_tokens": 20971520,
|
| 81 |
+
"shard_tokens": 100000000,
|
| 82 |
+
"tokenizer_threads": 8
|
| 83 |
+
},
|
| 84 |
+
"training": {
|
| 85 |
+
"depth": 12,
|
| 86 |
+
"scaling_params": 110100912,
|
| 87 |
+
"target_param_data_ratio": 11.25,
|
| 88 |
+
"max_seq_len": 8192,
|
| 89 |
+
"window_pattern": "L",
|
| 90 |
+
"device_batch_size": 4,
|
| 91 |
+
"total_batch_size": 524288,
|
| 92 |
+
"save_every": 500,
|
| 93 |
+
"eval_every": 250,
|
| 94 |
+
"eval_tokens": 2097152,
|
| 95 |
+
"core_metric_every": -1,
|
| 96 |
+
"sample_every": -1
|
| 97 |
+
},
|
| 98 |
+
"artifacts": {
|
| 99 |
+
"repo": "jbduran/think.nano"
|
| 100 |
+
},
|
| 101 |
+
"wandb": {
|
| 102 |
+
"entity": "jbduran-thinkingmachinesncsu",
|
| 103 |
+
"project": "think.nano",
|
| 104 |
+
"name": "think-d12-r11.25-ctx8192",
|
| 105 |
+
"group": "think-d12",
|
| 106 |
+
"tags": [
|
| 107 |
+
"think-dataset",
|
| 108 |
+
"d12",
|
| 109 |
+
"ratio11.25",
|
| 110 |
+
"ctx8192"
|
| 111 |
+
]
|
| 112 |
+
},
|
| 113 |
+
"config_fingerprint": "407a5074e0bf3730",
|
| 114 |
+
"artifact_path": "experiments/think-d12-r11.25-ctx8192"
|
| 115 |
+
},
|
| 116 |
+
"stage": "base",
|
| 117 |
+
"base_experiment_id": "think-d12-r11.25-ctx8192",
|
| 118 |
+
"parent_experiment_id": null,
|
| 119 |
+
"parent_checkpoint_step": null,
|
| 120 |
+
"config_fingerprint": "407a5074e0bf3730"
|
| 121 |
+
},
|
| 122 |
+
"device_batch_size": 4,
|
| 123 |
+
"max_seq_len": 8192,
|
| 124 |
+
"total_batch_size": 524288,
|
| 125 |
+
"dataloader_state_dict": {
|
| 126 |
+
"file_idx": 5,
|
| 127 |
+
"pos": 24336769,
|
| 128 |
+
"epoch": 1,
|
| 129 |
+
"pq_idx": 5,
|
| 130 |
+
"rg_idx": 24336769
|
| 131 |
+
},
|
| 132 |
+
"loop_state": {
|
| 133 |
+
"min_val_bpb": 1.1673074973592756,
|
| 134 |
+
"smooth_train_loss": 3.291130702382907,
|
| 135 |
+
"total_training_time": 3762.280524253845,
|
| 136 |
+
"stage_training_flops": 821336544903168000,
|
| 137 |
+
"inherited_parent_flops": 0.0,
|
| 138 |
+
"cumulative_pipeline_training_flops": 821336544903168000
|
| 139 |
+
}
|
| 140 |
+
}
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001500.json
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 1500,
|
| 3 |
+
"training_complete": false,
|
| 4 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 5 |
+
"val_bpb": 1.108596107059193,
|
| 6 |
+
"model_config": {
|
| 7 |
+
"sequence_len": 8192,
|
| 8 |
+
"vocab_size": 32768,
|
| 9 |
+
"n_layer": 12,
|
| 10 |
+
"n_head": 6,
|
| 11 |
+
"n_kv_head": 6,
|
| 12 |
+
"n_embd": 768,
|
| 13 |
+
"window_pattern": "L"
|
| 14 |
+
},
|
| 15 |
+
"user_config": {
|
| 16 |
+
"run": "think-d12-r11.25-ctx8192",
|
| 17 |
+
"wandb_run_id": "e3483a4b",
|
| 18 |
+
"wandb_group": "think-d12",
|
| 19 |
+
"wandb_tags": "think-dataset,d12,ratio11.25,ctx8192",
|
| 20 |
+
"device_type": "",
|
| 21 |
+
"fp8": false,
|
| 22 |
+
"fp8_recipe": "tensorwise",
|
| 23 |
+
"depth": 12,
|
| 24 |
+
"aspect_ratio": 64,
|
| 25 |
+
"head_dim": 128,
|
| 26 |
+
"max_seq_len": 8192,
|
| 27 |
+
"window_pattern": "L",
|
| 28 |
+
"num_iterations": -1,
|
| 29 |
+
"target_flops": -1.0,
|
| 30 |
+
"target_param_data_ratio": 11.25,
|
| 31 |
+
"device_batch_size": 4,
|
| 32 |
+
"total_batch_size": 524288,
|
| 33 |
+
"embedding_lr": 0.3,
|
| 34 |
+
"unembedding_lr": 0.008,
|
| 35 |
+
"weight_decay": 0.28,
|
| 36 |
+
"matrix_lr": 0.02,
|
| 37 |
+
"scalar_lr": 0.5,
|
| 38 |
+
"warmup_steps": 40,
|
| 39 |
+
"warmdown_ratio": 0.65,
|
| 40 |
+
"final_lr_frac": 0.05,
|
| 41 |
+
"resume_from_step": -1,
|
| 42 |
+
"pretokenized": true,
|
| 43 |
+
"data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data",
|
| 44 |
+
"tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer",
|
| 45 |
+
"pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok",
|
| 46 |
+
"checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints",
|
| 47 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 48 |
+
"experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json",
|
| 49 |
+
"tokenizer_fingerprint": "03c4f62e7a9d0c3b",
|
| 50 |
+
"git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847",
|
| 51 |
+
"seed": 42,
|
| 52 |
+
"eval_every": 250,
|
| 53 |
+
"eval_tokens": 2097152,
|
| 54 |
+
"core_metric_every": -1,
|
| 55 |
+
"core_metric_max_per_task": 500,
|
| 56 |
+
"sample_every": -1,
|
| 57 |
+
"save_every": 500,
|
| 58 |
+
"model_tag": "think-d12-r11.25-ctx8192",
|
| 59 |
+
"experiment": {
|
| 60 |
+
"schema_version": 1,
|
| 61 |
+
"stage": "base",
|
| 62 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 63 |
+
"dataset": {
|
| 64 |
+
"adapter": "parquet_shards",
|
| 65 |
+
"repo": "jbduran/think-dataset",
|
| 66 |
+
"revision": "main",
|
| 67 |
+
"validation_shard": 472,
|
| 68 |
+
"num_train_shards": 24,
|
| 69 |
+
"download_workers": 4
|
| 70 |
+
},
|
| 71 |
+
"tokenizer": {
|
| 72 |
+
"mode": "train",
|
| 73 |
+
"max_chars": 2000000000,
|
| 74 |
+
"doc_cap": 10000,
|
| 75 |
+
"vocab_size": 32768
|
| 76 |
+
},
|
| 77 |
+
"pretokenize": {
|
| 78 |
+
"enabled": true,
|
| 79 |
+
"slack": 1.03,
|
| 80 |
+
"val_tokens": 20971520,
|
| 81 |
+
"shard_tokens": 100000000,
|
| 82 |
+
"tokenizer_threads": 8
|
| 83 |
+
},
|
| 84 |
+
"training": {
|
| 85 |
+
"depth": 12,
|
| 86 |
+
"scaling_params": 110100912,
|
| 87 |
+
"target_param_data_ratio": 11.25,
|
| 88 |
+
"max_seq_len": 8192,
|
| 89 |
+
"window_pattern": "L",
|
| 90 |
+
"device_batch_size": 4,
|
| 91 |
+
"total_batch_size": 524288,
|
| 92 |
+
"save_every": 500,
|
| 93 |
+
"eval_every": 250,
|
| 94 |
+
"eval_tokens": 2097152,
|
| 95 |
+
"core_metric_every": -1,
|
| 96 |
+
"sample_every": -1
|
| 97 |
+
},
|
| 98 |
+
"artifacts": {
|
| 99 |
+
"repo": "jbduran/think.nano"
|
| 100 |
+
},
|
| 101 |
+
"wandb": {
|
| 102 |
+
"entity": "jbduran-thinkingmachinesncsu",
|
| 103 |
+
"project": "think.nano",
|
| 104 |
+
"name": "think-d12-r11.25-ctx8192",
|
| 105 |
+
"group": "think-d12",
|
| 106 |
+
"tags": [
|
| 107 |
+
"think-dataset",
|
| 108 |
+
"d12",
|
| 109 |
+
"ratio11.25",
|
| 110 |
+
"ctx8192"
|
| 111 |
+
]
|
| 112 |
+
},
|
| 113 |
+
"config_fingerprint": "407a5074e0bf3730",
|
| 114 |
+
"artifact_path": "experiments/think-d12-r11.25-ctx8192"
|
| 115 |
+
},
|
| 116 |
+
"stage": "base",
|
| 117 |
+
"base_experiment_id": "think-d12-r11.25-ctx8192",
|
| 118 |
+
"parent_experiment_id": null,
|
| 119 |
+
"parent_checkpoint_step": null,
|
| 120 |
+
"config_fingerprint": "407a5074e0bf3730"
|
| 121 |
+
},
|
| 122 |
+
"device_batch_size": 4,
|
| 123 |
+
"max_seq_len": 8192,
|
| 124 |
+
"total_batch_size": 524288,
|
| 125 |
+
"dataloader_state_dict": {
|
| 126 |
+
"file_idx": 7,
|
| 127 |
+
"pos": 86488769,
|
| 128 |
+
"epoch": 1,
|
| 129 |
+
"pq_idx": 7,
|
| 130 |
+
"rg_idx": 86488769
|
| 131 |
+
},
|
| 132 |
+
"loop_state": {
|
| 133 |
+
"min_val_bpb": 1.108596107059193,
|
| 134 |
+
"smooth_train_loss": 3.1279163553930784,
|
| 135 |
+
"total_training_time": 5663.0339615345,
|
| 136 |
+
"stage_training_flops": 1232004817354752000,
|
| 137 |
+
"inherited_parent_flops": 0.0,
|
| 138 |
+
"cumulative_pipeline_training_flops": 1232004817354752000
|
| 139 |
+
}
|
| 140 |
+
}
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002000.json
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 2000,
|
| 3 |
+
"training_complete": false,
|
| 4 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 5 |
+
"val_bpb": 1.0589838913914653,
|
| 6 |
+
"model_config": {
|
| 7 |
+
"sequence_len": 8192,
|
| 8 |
+
"vocab_size": 32768,
|
| 9 |
+
"n_layer": 12,
|
| 10 |
+
"n_head": 6,
|
| 11 |
+
"n_kv_head": 6,
|
| 12 |
+
"n_embd": 768,
|
| 13 |
+
"window_pattern": "L"
|
| 14 |
+
},
|
| 15 |
+
"user_config": {
|
| 16 |
+
"run": "think-d12-r11.25-ctx8192",
|
| 17 |
+
"wandb_run_id": "e3483a4b",
|
| 18 |
+
"wandb_group": "think-d12",
|
| 19 |
+
"wandb_tags": "think-dataset,d12,ratio11.25,ctx8192",
|
| 20 |
+
"device_type": "",
|
| 21 |
+
"fp8": false,
|
| 22 |
+
"fp8_recipe": "tensorwise",
|
| 23 |
+
"depth": 12,
|
| 24 |
+
"aspect_ratio": 64,
|
| 25 |
+
"head_dim": 128,
|
| 26 |
+
"max_seq_len": 8192,
|
| 27 |
+
"window_pattern": "L",
|
| 28 |
+
"num_iterations": -1,
|
| 29 |
+
"target_flops": -1.0,
|
| 30 |
+
"target_param_data_ratio": 11.25,
|
| 31 |
+
"device_batch_size": 4,
|
| 32 |
+
"total_batch_size": 524288,
|
| 33 |
+
"embedding_lr": 0.3,
|
| 34 |
+
"unembedding_lr": 0.008,
|
| 35 |
+
"weight_decay": 0.28,
|
| 36 |
+
"matrix_lr": 0.02,
|
| 37 |
+
"scalar_lr": 0.5,
|
| 38 |
+
"warmup_steps": 40,
|
| 39 |
+
"warmdown_ratio": 0.65,
|
| 40 |
+
"final_lr_frac": 0.05,
|
| 41 |
+
"resume_from_step": -1,
|
| 42 |
+
"pretokenized": true,
|
| 43 |
+
"data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data",
|
| 44 |
+
"tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer",
|
| 45 |
+
"pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok",
|
| 46 |
+
"checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints",
|
| 47 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 48 |
+
"experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json",
|
| 49 |
+
"tokenizer_fingerprint": "03c4f62e7a9d0c3b",
|
| 50 |
+
"git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847",
|
| 51 |
+
"seed": 42,
|
| 52 |
+
"eval_every": 250,
|
| 53 |
+
"eval_tokens": 2097152,
|
| 54 |
+
"core_metric_every": -1,
|
| 55 |
+
"core_metric_max_per_task": 500,
|
| 56 |
+
"sample_every": -1,
|
| 57 |
+
"save_every": 500,
|
| 58 |
+
"model_tag": "think-d12-r11.25-ctx8192",
|
| 59 |
+
"experiment": {
|
| 60 |
+
"schema_version": 1,
|
| 61 |
+
"stage": "base",
|
| 62 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 63 |
+
"dataset": {
|
| 64 |
+
"adapter": "parquet_shards",
|
| 65 |
+
"repo": "jbduran/think-dataset",
|
| 66 |
+
"revision": "main",
|
| 67 |
+
"validation_shard": 472,
|
| 68 |
+
"num_train_shards": 24,
|
| 69 |
+
"download_workers": 4
|
| 70 |
+
},
|
| 71 |
+
"tokenizer": {
|
| 72 |
+
"mode": "train",
|
| 73 |
+
"max_chars": 2000000000,
|
| 74 |
+
"doc_cap": 10000,
|
| 75 |
+
"vocab_size": 32768
|
| 76 |
+
},
|
| 77 |
+
"pretokenize": {
|
| 78 |
+
"enabled": true,
|
| 79 |
+
"slack": 1.03,
|
| 80 |
+
"val_tokens": 20971520,
|
| 81 |
+
"shard_tokens": 100000000,
|
| 82 |
+
"tokenizer_threads": 8
|
| 83 |
+
},
|
| 84 |
+
"training": {
|
| 85 |
+
"depth": 12,
|
| 86 |
+
"scaling_params": 110100912,
|
| 87 |
+
"target_param_data_ratio": 11.25,
|
| 88 |
+
"max_seq_len": 8192,
|
| 89 |
+
"window_pattern": "L",
|
| 90 |
+
"device_batch_size": 4,
|
| 91 |
+
"total_batch_size": 524288,
|
| 92 |
+
"save_every": 500,
|
| 93 |
+
"eval_every": 250,
|
| 94 |
+
"eval_tokens": 2097152,
|
| 95 |
+
"core_metric_every": -1,
|
| 96 |
+
"sample_every": -1
|
| 97 |
+
},
|
| 98 |
+
"artifacts": {
|
| 99 |
+
"repo": "jbduran/think.nano"
|
| 100 |
+
},
|
| 101 |
+
"wandb": {
|
| 102 |
+
"entity": "jbduran-thinkingmachinesncsu",
|
| 103 |
+
"project": "think.nano",
|
| 104 |
+
"name": "think-d12-r11.25-ctx8192",
|
| 105 |
+
"group": "think-d12",
|
| 106 |
+
"tags": [
|
| 107 |
+
"think-dataset",
|
| 108 |
+
"d12",
|
| 109 |
+
"ratio11.25",
|
| 110 |
+
"ctx8192"
|
| 111 |
+
]
|
| 112 |
+
},
|
| 113 |
+
"config_fingerprint": "407a5074e0bf3730",
|
| 114 |
+
"artifact_path": "experiments/think-d12-r11.25-ctx8192"
|
| 115 |
+
},
|
| 116 |
+
"stage": "base",
|
| 117 |
+
"base_experiment_id": "think-d12-r11.25-ctx8192",
|
| 118 |
+
"parent_experiment_id": null,
|
| 119 |
+
"parent_checkpoint_step": null,
|
| 120 |
+
"config_fingerprint": "407a5074e0bf3730"
|
| 121 |
+
},
|
| 122 |
+
"device_batch_size": 4,
|
| 123 |
+
"max_seq_len": 8192,
|
| 124 |
+
"total_batch_size": 524288,
|
| 125 |
+
"dataloader_state_dict": {
|
| 126 |
+
"file_idx": 10,
|
| 127 |
+
"pos": 48640769,
|
| 128 |
+
"epoch": 1,
|
| 129 |
+
"pq_idx": 10,
|
| 130 |
+
"rg_idx": 48640769
|
| 131 |
+
},
|
| 132 |
+
"loop_state": {
|
| 133 |
+
"min_val_bpb": 1.0589838913914653,
|
| 134 |
+
"smooth_train_loss": 3.116437961262795,
|
| 135 |
+
"total_training_time": 7566.1544008255005,
|
| 136 |
+
"stage_training_flops": 1642673089806336000,
|
| 137 |
+
"inherited_parent_flops": 0.0,
|
| 138 |
+
"cumulative_pipeline_training_flops": 1642673089806336000
|
| 139 |
+
}
|
| 140 |
+
}
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002362.json
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 2362,
|
| 3 |
+
"training_complete": true,
|
| 4 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 5 |
+
"val_bpb": 1.0395519592251896,
|
| 6 |
+
"model_config": {
|
| 7 |
+
"sequence_len": 8192,
|
| 8 |
+
"vocab_size": 32768,
|
| 9 |
+
"n_layer": 12,
|
| 10 |
+
"n_head": 6,
|
| 11 |
+
"n_kv_head": 6,
|
| 12 |
+
"n_embd": 768,
|
| 13 |
+
"window_pattern": "L"
|
| 14 |
+
},
|
| 15 |
+
"user_config": {
|
| 16 |
+
"run": "think-d12-r11.25-ctx8192",
|
| 17 |
+
"wandb_run_id": "e3483a4b",
|
| 18 |
+
"wandb_group": "think-d12",
|
| 19 |
+
"wandb_tags": "think-dataset,d12,ratio11.25,ctx8192",
|
| 20 |
+
"device_type": "",
|
| 21 |
+
"fp8": false,
|
| 22 |
+
"fp8_recipe": "tensorwise",
|
| 23 |
+
"depth": 12,
|
| 24 |
+
"aspect_ratio": 64,
|
| 25 |
+
"head_dim": 128,
|
| 26 |
+
"max_seq_len": 8192,
|
| 27 |
+
"window_pattern": "L",
|
| 28 |
+
"num_iterations": -1,
|
| 29 |
+
"target_flops": -1.0,
|
| 30 |
+
"target_param_data_ratio": 11.25,
|
| 31 |
+
"device_batch_size": 4,
|
| 32 |
+
"total_batch_size": 524288,
|
| 33 |
+
"embedding_lr": 0.3,
|
| 34 |
+
"unembedding_lr": 0.008,
|
| 35 |
+
"weight_decay": 0.28,
|
| 36 |
+
"matrix_lr": 0.02,
|
| 37 |
+
"scalar_lr": 0.5,
|
| 38 |
+
"warmup_steps": 40,
|
| 39 |
+
"warmdown_ratio": 0.65,
|
| 40 |
+
"final_lr_frac": 0.05,
|
| 41 |
+
"resume_from_step": 2000,
|
| 42 |
+
"pretokenized": true,
|
| 43 |
+
"data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data",
|
| 44 |
+
"tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer",
|
| 45 |
+
"pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok",
|
| 46 |
+
"checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints",
|
| 47 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 48 |
+
"experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json",
|
| 49 |
+
"tokenizer_fingerprint": "03c4f62e7a9d0c3b",
|
| 50 |
+
"git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847",
|
| 51 |
+
"seed": 42,
|
| 52 |
+
"eval_every": 250,
|
| 53 |
+
"eval_tokens": 2097152,
|
| 54 |
+
"core_metric_every": -1,
|
| 55 |
+
"core_metric_max_per_task": 500,
|
| 56 |
+
"sample_every": -1,
|
| 57 |
+
"save_every": 500,
|
| 58 |
+
"model_tag": "think-d12-r11.25-ctx8192",
|
| 59 |
+
"experiment": {
|
| 60 |
+
"schema_version": 1,
|
| 61 |
+
"stage": "base",
|
| 62 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 63 |
+
"dataset": {
|
| 64 |
+
"adapter": "parquet_shards",
|
| 65 |
+
"repo": "jbduran/think-dataset",
|
| 66 |
+
"revision": "main",
|
| 67 |
+
"validation_shard": 472,
|
| 68 |
+
"num_train_shards": 24,
|
| 69 |
+
"download_workers": 4
|
| 70 |
+
},
|
| 71 |
+
"tokenizer": {
|
| 72 |
+
"mode": "train",
|
| 73 |
+
"max_chars": 2000000000,
|
| 74 |
+
"doc_cap": 10000,
|
| 75 |
+
"vocab_size": 32768
|
| 76 |
+
},
|
| 77 |
+
"pretokenize": {
|
| 78 |
+
"enabled": true,
|
| 79 |
+
"slack": 1.03,
|
| 80 |
+
"val_tokens": 20971520,
|
| 81 |
+
"shard_tokens": 100000000,
|
| 82 |
+
"tokenizer_threads": 8
|
| 83 |
+
},
|
| 84 |
+
"training": {
|
| 85 |
+
"depth": 12,
|
| 86 |
+
"scaling_params": 110100912,
|
| 87 |
+
"target_param_data_ratio": 11.25,
|
| 88 |
+
"max_seq_len": 8192,
|
| 89 |
+
"window_pattern": "L",
|
| 90 |
+
"device_batch_size": 4,
|
| 91 |
+
"total_batch_size": 524288,
|
| 92 |
+
"save_every": 500,
|
| 93 |
+
"eval_every": 250,
|
| 94 |
+
"eval_tokens": 2097152,
|
| 95 |
+
"core_metric_every": -1,
|
| 96 |
+
"sample_every": -1
|
| 97 |
+
},
|
| 98 |
+
"artifacts": {
|
| 99 |
+
"repo": "jbduran/think.nano"
|
| 100 |
+
},
|
| 101 |
+
"wandb": {
|
| 102 |
+
"entity": "jbduran-thinkingmachinesncsu",
|
| 103 |
+
"project": "think.nano",
|
| 104 |
+
"name": "think-d12-r11.25-ctx8192",
|
| 105 |
+
"group": "think-d12",
|
| 106 |
+
"tags": [
|
| 107 |
+
"think-dataset",
|
| 108 |
+
"d12",
|
| 109 |
+
"ratio11.25",
|
| 110 |
+
"ctx8192"
|
| 111 |
+
]
|
| 112 |
+
},
|
| 113 |
+
"config_fingerprint": "407a5074e0bf3730",
|
| 114 |
+
"artifact_path": "experiments/think-d12-r11.25-ctx8192"
|
| 115 |
+
},
|
| 116 |
+
"stage": "base",
|
| 117 |
+
"base_experiment_id": "think-d12-r11.25-ctx8192",
|
| 118 |
+
"parent_experiment_id": null,
|
| 119 |
+
"parent_checkpoint_step": null,
|
| 120 |
+
"config_fingerprint": "407a5074e0bf3730"
|
| 121 |
+
},
|
| 122 |
+
"device_batch_size": 4,
|
| 123 |
+
"max_seq_len": 8192,
|
| 124 |
+
"total_batch_size": 524288,
|
| 125 |
+
"dataloader_state_dict": {
|
| 126 |
+
"file_idx": 12,
|
| 127 |
+
"pos": 38471586,
|
| 128 |
+
"epoch": 1,
|
| 129 |
+
"pq_idx": 12,
|
| 130 |
+
"rg_idx": 38471586
|
| 131 |
+
},
|
| 132 |
+
"loop_state": {
|
| 133 |
+
"min_val_bpb": 1.0395519592251896,
|
| 134 |
+
"smooth_train_loss": 3.0155535492313272,
|
| 135 |
+
"total_training_time": 9008.03685593605,
|
| 136 |
+
"stage_training_flops": 1939996919061282816,
|
| 137 |
+
"inherited_parent_flops": 0.0,
|
| 138 |
+
"cumulative_pipeline_training_flops": 1939996919061282816
|
| 139 |
+
}
|
| 140 |
+
}
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_000500.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:51922864ab7ccea4cd40458bb60898167d02696982c1e4c0de684995e5f26290
|
| 3 |
+
size 792761690
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001000.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c8c23459c7bef825a21a0236f2b9e2efba66c8b427c78a7a6cca852df957fd0e
|
| 3 |
+
size 792761690
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001500.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5dd6cc545ba54787b342e190d75e857872d1964f10a4032ab0d49640012f84b7
|
| 3 |
+
size 792761690
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002000.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d6e6c50126914bc4d20f9db6222891ee9ee61c634634b023a13e5f1e583e9403
|
| 3 |
+
size 792761690
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002362.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:43a07973a6d23613f28d1b43623e06bba393623fd88452e1c495bf26aa9ac4d6
|
| 3 |
+
size 792761690
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_000500_rank0.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f402cf80bf9d4b917ae2a99240b0ea34cd842f025927c148bccf384ce9b744b8
|
| 3 |
+
size 1246165357
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001000_rank0.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9ef8f97be5f4890871d5586afd42bc946b9058ae996deaa83c14c8f71610deb9
|
| 3 |
+
size 1246165357
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001500_rank0.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:df9bf4a59d46c126bd65f5cdc6de8fb531c492fabf70f9db22a5ac27bd08dd2f
|
| 3 |
+
size 1246165357
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002000_rank0.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f95fc4fe7c0a841f3e443102d00495e4f4305ea955ad2b81ffd3ad8865cbd34e
|
| 3 |
+
size 1246165357
|
experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002362_rank0.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9e1464e581c921562375dd5957b76dad9fe37d04a17e9a021a7876026e624044
|
| 3 |
+
size 1246165357
|
experiments/think-d12-r11.25-ctx8192/config.json
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"schema_version": 1,
|
| 3 |
+
"stage": "base",
|
| 4 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 5 |
+
"dataset": {
|
| 6 |
+
"adapter": "parquet_shards",
|
| 7 |
+
"repo": "jbduran/think-dataset",
|
| 8 |
+
"revision": "main",
|
| 9 |
+
"validation_shard": 472,
|
| 10 |
+
"num_train_shards": 24,
|
| 11 |
+
"download_workers": 4
|
| 12 |
+
},
|
| 13 |
+
"tokenizer": {
|
| 14 |
+
"mode": "train",
|
| 15 |
+
"max_chars": 2000000000,
|
| 16 |
+
"doc_cap": 10000,
|
| 17 |
+
"vocab_size": 32768
|
| 18 |
+
},
|
| 19 |
+
"pretokenize": {
|
| 20 |
+
"enabled": true,
|
| 21 |
+
"slack": 1.03,
|
| 22 |
+
"val_tokens": 20971520,
|
| 23 |
+
"shard_tokens": 100000000,
|
| 24 |
+
"tokenizer_threads": 8
|
| 25 |
+
},
|
| 26 |
+
"training": {
|
| 27 |
+
"depth": 12,
|
| 28 |
+
"scaling_params": 110100912,
|
| 29 |
+
"target_param_data_ratio": 11.25,
|
| 30 |
+
"max_seq_len": 8192,
|
| 31 |
+
"window_pattern": "L",
|
| 32 |
+
"device_batch_size": 4,
|
| 33 |
+
"total_batch_size": 524288,
|
| 34 |
+
"save_every": 500,
|
| 35 |
+
"eval_every": 250,
|
| 36 |
+
"eval_tokens": 2097152,
|
| 37 |
+
"core_metric_every": -1,
|
| 38 |
+
"sample_every": -1
|
| 39 |
+
},
|
| 40 |
+
"artifacts": {
|
| 41 |
+
"repo": "jbduran/think.nano"
|
| 42 |
+
},
|
| 43 |
+
"wandb": {
|
| 44 |
+
"entity": "jbduran-thinkingmachinesncsu",
|
| 45 |
+
"project": "think.nano",
|
| 46 |
+
"name": "think-d12-r11.25-ctx8192",
|
| 47 |
+
"group": "think-d12",
|
| 48 |
+
"tags": [
|
| 49 |
+
"think-dataset",
|
| 50 |
+
"d12",
|
| 51 |
+
"ratio11.25",
|
| 52 |
+
"ctx8192"
|
| 53 |
+
]
|
| 54 |
+
},
|
| 55 |
+
"config_fingerprint": "407a5074e0bf3730",
|
| 56 |
+
"artifact_path": "experiments/think-d12-r11.25-ctx8192"
|
| 57 |
+
}
|
experiments/think-d12-r11.25-ctx8192/evals/core.json
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "base_model (step 2362)",
|
| 3 |
+
"step": 2362,
|
| 4 |
+
"bpb": {},
|
| 5 |
+
"core_metric": 0.07251341817663091,
|
| 6 |
+
"core_results": {
|
| 7 |
+
"hellaswag_zeroshot": 0.2757418751716614,
|
| 8 |
+
"jeopardy": 0.0009447330958209932,
|
| 9 |
+
"bigbench_qa_wikidata": 0.0832144096493721,
|
| 10 |
+
"arc_easy": 0.3194444477558136,
|
| 11 |
+
"arc_challenge": 0.21501706540584564,
|
| 12 |
+
"copa": 0.5099999904632568,
|
| 13 |
+
"commonsense_qa": 0.31285831332206726,
|
| 14 |
+
"piqa": 0.5331882238388062,
|
| 15 |
+
"openbook_qa": 0.24800001084804535,
|
| 16 |
+
"lambada_openai": 0.23423248529434204,
|
| 17 |
+
"hellaswag": 0.2802230417728424,
|
| 18 |
+
"winograd": 0.5714285969734192,
|
| 19 |
+
"winogrande": 0.4956590235233307,
|
| 20 |
+
"bigbench_dyck_languages": 0.10200000554323196,
|
| 21 |
+
"agi_eval_lsat_ar": 0.260869562625885,
|
| 22 |
+
"bigbench_cs_algorithms": 0.41969695687294006,
|
| 23 |
+
"bigbench_operators": 0.07619047909975052,
|
| 24 |
+
"bigbench_repeat_copy_logic": 0.0,
|
| 25 |
+
"squad": 0.024030273780226707,
|
| 26 |
+
"coqa": 0.0821746215224266,
|
| 27 |
+
"boolq": 0.5590214133262634,
|
| 28 |
+
"bigbench_language_identification": 0.2524999976158142
|
| 29 |
+
},
|
| 30 |
+
"centered_results": {
|
| 31 |
+
"hellaswag_zeroshot": 0.034322500228881836,
|
| 32 |
+
"jeopardy": 0.0009447330958209932,
|
| 33 |
+
"bigbench_qa_wikidata": 0.0832144096493721,
|
| 34 |
+
"arc_easy": 0.09259259700775146,
|
| 35 |
+
"arc_challenge": -0.04664391279220581,
|
| 36 |
+
"copa": 0.019999980926513672,
|
| 37 |
+
"commonsense_qa": 0.14107289165258405,
|
| 38 |
+
"piqa": 0.0663764476776123,
|
| 39 |
+
"openbook_qa": -0.002666652202606201,
|
| 40 |
+
"lambada_openai": 0.23423248529434204,
|
| 41 |
+
"hellaswag": 0.04029738903045654,
|
| 42 |
+
"winograd": 0.14285719394683838,
|
| 43 |
+
"winogrande": -0.008681952953338623,
|
| 44 |
+
"bigbench_dyck_languages": 0.10200000554323196,
|
| 45 |
+
"agi_eval_lsat_ar": 0.07608695328235625,
|
| 46 |
+
"bigbench_cs_algorithms": 0.41969695687294006,
|
| 47 |
+
"bigbench_operators": 0.07619047909975052,
|
| 48 |
+
"bigbench_repeat_copy_logic": 0.0,
|
| 49 |
+
"squad": 0.024030273780226707,
|
| 50 |
+
"coqa": 0.0821746215224266,
|
| 51 |
+
"boolq": -0.1604699649308857,
|
| 52 |
+
"bigbench_language_identification": 0.177667764153811
|
| 53 |
+
},
|
| 54 |
+
"conditioned_samples": [],
|
| 55 |
+
"unconditioned_samples": []
|
| 56 |
+
}
|
experiments/think-d12-r11.25-ctx8192/evals/samples.json
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "base_model (step 2362)",
|
| 3 |
+
"step": 2362,
|
| 4 |
+
"bpb": {},
|
| 5 |
+
"core_metric": null,
|
| 6 |
+
"core_results": null,
|
| 7 |
+
"centered_results": null,
|
| 8 |
+
"conditioned_samples": [
|
| 9 |
+
{
|
| 10 |
+
"prompt": "The capital of France is",
|
| 11 |
+
"text": "<|bos|>The capital of France is not yet fully developed. The capital of the United States is not yet fully developed"
|
| 12 |
+
},
|
| 13 |
+
{
|
| 14 |
+
"prompt": "The chemical symbol of gold is",
|
| 15 |
+
"text": "<|bos|>The chemical symbol of gold is the symbol of the gold, and the symbol of the silver. The gold is"
|
| 16 |
+
},
|
| 17 |
+
{
|
| 18 |
+
"prompt": "If yesterday was Friday, then tomorrow will be",
|
| 19 |
+
"text": "<|bos|>If yesterday was Friday, then tomorrow will be the day of the week. \n\nThe day of the week is the same as"
|
| 20 |
+
},
|
| 21 |
+
{
|
| 22 |
+
"prompt": "The opposite of hot is",
|
| 23 |
+
"text": "<|bos|>The opposite of hot is the same as hot. \n\nThe hot is the same as hot. \n\nThe"
|
| 24 |
+
},
|
| 25 |
+
{
|
| 26 |
+
"prompt": "The planets of the solar system are:",
|
| 27 |
+
"text": "<|bos|>The planets of the solar system are: \n\n1. The sun, 2. The moon, 3. The"
|
| 28 |
+
},
|
| 29 |
+
{
|
| 30 |
+
"prompt": "My favorite color is",
|
| 31 |
+
"text": "<|bos|>My favorite color is the color of the skin of the face, and the color of the skin."
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"prompt": "If 5*x + 3 = 13, then x is",
|
| 35 |
+
"text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the"
|
| 36 |
+
}
|
| 37 |
+
],
|
| 38 |
+
"unconditioned_samples": [
|
| 39 |
+
"<|bos|>ALIENS. A concern called industrial, in which all miners of ability for useful labor were engaged in obtaining industrial materials for making snuffers. Although it would be a rude and untenable enterprise to make different classes of miners dispose of goods for profit, different miners differing between the quality of the material misspelled and its quantity, it always staggers the mind with the idea of the matter which it concerns.\u00b9 Four or five shopkeepers are seen at so many tradeshops in town near together in towns and villages. \n\nMoney is better paid to supply the needs of skilled men than it is in overcrowded",
|
| 40 |
+
"<|bos|>37374.31 617.471.11 379.75 \n\nSwinburne, Jes. 10, 335.\n\nStatistical Index. \n\nCoates, erance, 1333. \n\nPurple Debenture, 1460.\n\nGenetic Index. \n\nSwinfenning, 51858.29, 1319. \n\nFree Presses. \n\nRidgens-Pantrepous, 25. \n\nSprayl-Power, regular exercise, 1700. \n\nPreachers and Teachers of the Schools, 401\u20132. \n\n",
|
| 41 |
+
"<|bos|>URE FOOD AND THE BODY' \n\nIn such a commonwealth Siamese readers might find in Kumber's Essays, or Mabon's Vol. of Tobit and St. Jerome's Lives, a sound, a clear and satisfactory explanation of this phrase. May Lady Cassius inform the reverend Society from which this paragraph is borrowed that in this hour of peril men and women may \"paint to him,\" and bewail Rest and Treatment. In both of these circles there are three main meanings attached to the phrase. One is, with the excessive reference in hotel-keepers, donkeys, or hares; another, with alligators",
|
| 42 |
+
"<|bos|>HENRY MART 140. \n\nLeblay, Mr. De Martyn's invention of music, 4. IX.\n\nTRANSLATOR'S NOTE.-Send forth a translation of the Notes which were received by me, translated from the Musical \n\nCommission's Calendar.\n\nHis performance we cannot altogether estimate, but St. Columba gave us confirmation of his inventions, 30. XXVIIii, 18. Eh, What (Georgics, I, 177); 'Slightest book that ever was written' (Sonn., lies 28\u00bd), a work of high merit read with honour,\n\nQu\u00e6rese",
|
| 43 |
+
"<|bos|>Harvard School, IV Department of Education, 1843-1972. \n\n2 Henry State League, LL. concerning Courses in Medicine and the Arts, pp. 22 et seq.\n\nHistory of the Monroe Doctrine, by one who has visited Europe, compiled from European Authorities.\n\nNew York: N. Y. \n\nExaminer, Vol. XXXI, \n\nApril, 1917, p. 88.\n\nPamphlet on Scien tific Methods of Education (\"Outlines of the Maladies and Defects of the Methods of Industrial Society,\" by Dr. Jevons). \n\nNew York, September and October, 19",
|
| 44 |
+
"<|bos|>The Bird reflects upon. his. \n\nThe Clipper. \n\nVapour.\n\nUntil within a few weeks the admission of the truth to our beloved Bird was fatal to that race, little cared for neither in her recorded history nor since they married, and still less as regards her character. Her reign ended; and when she died only after a few months good for nothing the country felt herself well restored to health. She began now to see her way. Our dear bird became as dear to her as the Christian mother; she began to see her way clearer to her senses; and as her thoughts turned, and freedom fell back, she",
|
| 45 |
+
"<|bos|>Army of the Cumberland and Arkansas Army, and an Army of the Potomac under the command of Martin Robertson.\n\nHEADQUARTERS CAMP THIRDQUARTERS, THIRD BRIG 1ST BRIG 1ST BRIG 1ST BRIG 1ST BRIG \n\n6 8 8 \n\nMCLQUERISHER'S STATION, 9 P.M. \n\nMY ARMY, CAL. \n\nEnlarged with orders by the War Department.\n\nHeadquarters Camp War Department, Clope Ridge, Va., September 15, 1864. \n\n6 P.M The Confederates tend S. M. Camp are in the Confederate service hospital at N. C.",
|
| 46 |
+
"<|bos|>the eightieth year of his age.\n\nI had never been conferring, like the palette and the gallens at which I used to sit. never had thought it wrong to give the faintest hint of this prudery, which I trust is always requested of the upholder at his housekeeping, as shall appear by the order and directions accompanying it. So while I was listening to the wise old voice of the tender bride calling alone in her measure that nonsense of patriarchal impiety. It was all the more gratifying when I heard that the Governor of Shetton is now apparently labouring in the same breath, when he speaks"
|
| 47 |
+
]
|
| 48 |
+
}
|
experiments/think-d12-r11.25-ctx8192/evals/val_bpb.json
ADDED
|
@@ -0,0 +1,174 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "base_model (step 2362)",
|
| 3 |
+
"step": 2362,
|
| 4 |
+
"bpb": {
|
| 5 |
+
"val_per_position": [
|
| 6 |
+
{
|
| 7 |
+
"start": 0,
|
| 8 |
+
"end": 256,
|
| 9 |
+
"bpb": 1.1345637945630735
|
| 10 |
+
},
|
| 11 |
+
{
|
| 12 |
+
"start": 256,
|
| 13 |
+
"end": 512,
|
| 14 |
+
"bpb": 1.0678589586934193
|
| 15 |
+
},
|
| 16 |
+
{
|
| 17 |
+
"start": 512,
|
| 18 |
+
"end": 768,
|
| 19 |
+
"bpb": 1.0513444442729074
|
| 20 |
+
},
|
| 21 |
+
{
|
| 22 |
+
"start": 768,
|
| 23 |
+
"end": 1024,
|
| 24 |
+
"bpb": 1.0420207610099703
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"start": 1024,
|
| 28 |
+
"end": 1280,
|
| 29 |
+
"bpb": 1.0363256199642377
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"start": 1280,
|
| 33 |
+
"end": 1536,
|
| 34 |
+
"bpb": 1.0293503892901525
|
| 35 |
+
},
|
| 36 |
+
{
|
| 37 |
+
"start": 1536,
|
| 38 |
+
"end": 1792,
|
| 39 |
+
"bpb": 1.0270513457476949
|
| 40 |
+
},
|
| 41 |
+
{
|
| 42 |
+
"start": 1792,
|
| 43 |
+
"end": 2048,
|
| 44 |
+
"bpb": 1.0255301775348173
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"start": 2048,
|
| 48 |
+
"end": 2304,
|
| 49 |
+
"bpb": 1.0171632048394768
|
| 50 |
+
},
|
| 51 |
+
{
|
| 52 |
+
"start": 2304,
|
| 53 |
+
"end": 2560,
|
| 54 |
+
"bpb": 1.0150940036004366
|
| 55 |
+
},
|
| 56 |
+
{
|
| 57 |
+
"start": 2560,
|
| 58 |
+
"end": 2816,
|
| 59 |
+
"bpb": 1.0132597834490384
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"start": 2816,
|
| 63 |
+
"end": 3072,
|
| 64 |
+
"bpb": 1.0154303287433495
|
| 65 |
+
},
|
| 66 |
+
{
|
| 67 |
+
"start": 3072,
|
| 68 |
+
"end": 3328,
|
| 69 |
+
"bpb": 1.0125085420227948
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"start": 3328,
|
| 73 |
+
"end": 3584,
|
| 74 |
+
"bpb": 1.0117220799135607
|
| 75 |
+
},
|
| 76 |
+
{
|
| 77 |
+
"start": 3584,
|
| 78 |
+
"end": 3840,
|
| 79 |
+
"bpb": 1.0093052242865306
|
| 80 |
+
},
|
| 81 |
+
{
|
| 82 |
+
"start": 3840,
|
| 83 |
+
"end": 4096,
|
| 84 |
+
"bpb": 1.0067290549863526
|
| 85 |
+
},
|
| 86 |
+
{
|
| 87 |
+
"start": 4096,
|
| 88 |
+
"end": 4352,
|
| 89 |
+
"bpb": 1.007627760298138
|
| 90 |
+
},
|
| 91 |
+
{
|
| 92 |
+
"start": 4352,
|
| 93 |
+
"end": 4608,
|
| 94 |
+
"bpb": 1.005475773581573
|
| 95 |
+
},
|
| 96 |
+
{
|
| 97 |
+
"start": 4608,
|
| 98 |
+
"end": 4864,
|
| 99 |
+
"bpb": 1.0011348016292028
|
| 100 |
+
},
|
| 101 |
+
{
|
| 102 |
+
"start": 4864,
|
| 103 |
+
"end": 5120,
|
| 104 |
+
"bpb": 1.0025369118565095
|
| 105 |
+
},
|
| 106 |
+
{
|
| 107 |
+
"start": 5120,
|
| 108 |
+
"end": 5376,
|
| 109 |
+
"bpb": 0.9978363303279962
|
| 110 |
+
},
|
| 111 |
+
{
|
| 112 |
+
"start": 5376,
|
| 113 |
+
"end": 5632,
|
| 114 |
+
"bpb": 0.9936016054109623
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"start": 5632,
|
| 118 |
+
"end": 5888,
|
| 119 |
+
"bpb": 0.9930789448128515
|
| 120 |
+
},
|
| 121 |
+
{
|
| 122 |
+
"start": 5888,
|
| 123 |
+
"end": 6144,
|
| 124 |
+
"bpb": 0.988120986726172
|
| 125 |
+
},
|
| 126 |
+
{
|
| 127 |
+
"start": 6144,
|
| 128 |
+
"end": 6400,
|
| 129 |
+
"bpb": 0.9878455863729801
|
| 130 |
+
},
|
| 131 |
+
{
|
| 132 |
+
"start": 6400,
|
| 133 |
+
"end": 6656,
|
| 134 |
+
"bpb": 0.988322568304946
|
| 135 |
+
},
|
| 136 |
+
{
|
| 137 |
+
"start": 6656,
|
| 138 |
+
"end": 6912,
|
| 139 |
+
"bpb": 0.9895607000315525
|
| 140 |
+
},
|
| 141 |
+
{
|
| 142 |
+
"start": 6912,
|
| 143 |
+
"end": 7168,
|
| 144 |
+
"bpb": 0.9924135354945043
|
| 145 |
+
},
|
| 146 |
+
{
|
| 147 |
+
"start": 7168,
|
| 148 |
+
"end": 7424,
|
| 149 |
+
"bpb": 0.9887797597227126
|
| 150 |
+
},
|
| 151 |
+
{
|
| 152 |
+
"start": 7424,
|
| 153 |
+
"end": 7680,
|
| 154 |
+
"bpb": 0.9863644464805957
|
| 155 |
+
},
|
| 156 |
+
{
|
| 157 |
+
"start": 7680,
|
| 158 |
+
"end": 7936,
|
| 159 |
+
"bpb": 0.9852729515784135
|
| 160 |
+
},
|
| 161 |
+
{
|
| 162 |
+
"start": 7936,
|
| 163 |
+
"end": 8192,
|
| 164 |
+
"bpb": 0.9838855089135268
|
| 165 |
+
}
|
| 166 |
+
],
|
| 167 |
+
"val": 1.0127322889400117
|
| 168 |
+
},
|
| 169 |
+
"core_metric": null,
|
| 170 |
+
"core_results": null,
|
| 171 |
+
"centered_results": null,
|
| 172 |
+
"conditioned_samples": [],
|
| 173 |
+
"unconditioned_samples": []
|
| 174 |
+
}
|
experiments/think-d12-r11.25-ctx8192/run.json
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 3 |
+
"stage": "base",
|
| 4 |
+
"base_experiment_id": "think-d12-r11.25-ctx8192",
|
| 5 |
+
"parent_experiment_id": null,
|
| 6 |
+
"parent_checkpoint_step": null,
|
| 7 |
+
"config_fingerprint": "407a5074e0bf3730",
|
| 8 |
+
"wandb_run_id": "e3483a4b",
|
| 9 |
+
"created_at": 1783693257
|
| 10 |
+
}
|
experiments/think-d12-r11.25-ctx8192/summary.json
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 3 |
+
"stage": "base",
|
| 4 |
+
"base_experiment_id": "think-d12-r11.25-ctx8192",
|
| 5 |
+
"parent_experiment_id": null,
|
| 6 |
+
"parent_checkpoint_step": null,
|
| 7 |
+
"dataset": "jbduran/think-dataset",
|
| 8 |
+
"dataset_revision": "main",
|
| 9 |
+
"step": 2362,
|
| 10 |
+
"depth": 12,
|
| 11 |
+
"target_param_data_ratio": 11.25,
|
| 12 |
+
"training_tokens": 1238368256,
|
| 13 |
+
"final_sampled_val_bpb": 1.0395519592251896,
|
| 14 |
+
"minimum_sampled_val_bpb": 1.0395519592251896,
|
| 15 |
+
"full_val_bpb": 1.0127322889400117,
|
| 16 |
+
"core_metric": null,
|
| 17 |
+
"centered_results": null,
|
| 18 |
+
"conditioned_samples": [
|
| 19 |
+
{
|
| 20 |
+
"prompt": "The capital of France is",
|
| 21 |
+
"text": "<|bos|>The capital of France is not yet fully developed. The capital of the United States is not yet fully developed"
|
| 22 |
+
},
|
| 23 |
+
{
|
| 24 |
+
"prompt": "The chemical symbol of gold is",
|
| 25 |
+
"text": "<|bos|>The chemical symbol of gold is the symbol of the gold, and the symbol of the silver. The gold is"
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
"prompt": "If yesterday was Friday, then tomorrow will be",
|
| 29 |
+
"text": "<|bos|>If yesterday was Friday, then tomorrow will be the day of the week. \n\nThe day of the week is the same as"
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"prompt": "The opposite of hot is",
|
| 33 |
+
"text": "<|bos|>The opposite of hot is the same as hot. \n\nThe hot is the same as hot. \n\nThe"
|
| 34 |
+
},
|
| 35 |
+
{
|
| 36 |
+
"prompt": "The planets of the solar system are:",
|
| 37 |
+
"text": "<|bos|>The planets of the solar system are: \n\n1. The sun, 2. The moon, 3. The"
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"prompt": "My favorite color is",
|
| 41 |
+
"text": "<|bos|>My favorite color is the color of the skin of the face, and the color of the skin."
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"prompt": "If 5*x + 3 = 13, then x is",
|
| 45 |
+
"text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the"
|
| 46 |
+
}
|
| 47 |
+
],
|
| 48 |
+
"unconditioned_samples": [
|
| 49 |
+
"<|bos|>ALIENS. A concern called industrial, in which all miners of ability for useful labor were engaged in obtaining industrial materials for making snuffers. Although it would be a rude and untenable enterprise to make different classes of miners dispose of goods for profit, different miners differing between the quality of the material misspelled and its quantity, it always staggers the mind with the idea of the matter which it concerns.\u00b9 Four or five shopkeepers are seen at so many tradeshops in town near together in towns and villages. \n\nMoney is better paid to supply the needs of skilled men than it is in overcrowded",
|
| 50 |
+
"<|bos|>37374.31 617.471.11 379.75 \n\nSwinburne, Jes. 10, 335.\n\nStatistical Index. \n\nCoates, erance, 1333. \n\nPurple Debenture, 1460.\n\nGenetic Index. \n\nSwinfenning, 51858.29, 1319. \n\nFree Presses. \n\nRidgens-Pantrepous, 25. \n\nSprayl-Power, regular exercise, 1700. \n\nPreachers and Teachers of the Schools, 401\u20132. \n\n",
|
| 51 |
+
"<|bos|>URE FOOD AND THE BODY' \n\nIn such a commonwealth Siamese readers might find in Kumber's Essays, or Mabon's Vol. of Tobit and St. Jerome's Lives, a sound, a clear and satisfactory explanation of this phrase. May Lady Cassius inform the reverend Society from which this paragraph is borrowed that in this hour of peril men and women may \"paint to him,\" and bewail Rest and Treatment. In both of these circles there are three main meanings attached to the phrase. One is, with the excessive reference in hotel-keepers, donkeys, or hares; another, with alligators",
|
| 52 |
+
"<|bos|>HENRY MART 140. \n\nLeblay, Mr. De Martyn's invention of music, 4. IX.\n\nTRANSLATOR'S NOTE.-Send forth a translation of the Notes which were received by me, translated from the Musical \n\nCommission's Calendar.\n\nHis performance we cannot altogether estimate, but St. Columba gave us confirmation of his inventions, 30. XXVIIii, 18. Eh, What (Georgics, I, 177); 'Slightest book that ever was written' (Sonn., lies 28\u00bd), a work of high merit read with honour,\n\nQu\u00e6rese",
|
| 53 |
+
"<|bos|>Harvard School, IV Department of Education, 1843-1972. \n\n2 Henry State League, LL. concerning Courses in Medicine and the Arts, pp. 22 et seq.\n\nHistory of the Monroe Doctrine, by one who has visited Europe, compiled from European Authorities.\n\nNew York: N. Y. \n\nExaminer, Vol. XXXI, \n\nApril, 1917, p. 88.\n\nPamphlet on Scien tific Methods of Education (\"Outlines of the Maladies and Defects of the Methods of Industrial Society,\" by Dr. Jevons). \n\nNew York, September and October, 19",
|
| 54 |
+
"<|bos|>The Bird reflects upon. his. \n\nThe Clipper. \n\nVapour.\n\nUntil within a few weeks the admission of the truth to our beloved Bird was fatal to that race, little cared for neither in her recorded history nor since they married, and still less as regards her character. Her reign ended; and when she died only after a few months good for nothing the country felt herself well restored to health. She began now to see her way. Our dear bird became as dear to her as the Christian mother; she began to see her way clearer to her senses; and as her thoughts turned, and freedom fell back, she",
|
| 55 |
+
"<|bos|>Army of the Cumberland and Arkansas Army, and an Army of the Potomac under the command of Martin Robertson.\n\nHEADQUARTERS CAMP THIRDQUARTERS, THIRD BRIG 1ST BRIG 1ST BRIG 1ST BRIG 1ST BRIG \n\n6 8 8 \n\nMCLQUERISHER'S STATION, 9 P.M. \n\nMY ARMY, CAL. \n\nEnlarged with orders by the War Department.\n\nHeadquarters Camp War Department, Clope Ridge, Va., September 15, 1864. \n\n6 P.M The Confederates tend S. M. Camp are in the Confederate service hospital at N. C.",
|
| 56 |
+
"<|bos|>the eightieth year of his age.\n\nI had never been conferring, like the palette and the gallens at which I used to sit. never had thought it wrong to give the faintest hint of this prudery, which I trust is always requested of the upholder at his housekeeping, as shall appear by the order and directions accompanying it. So while I was listening to the wise old voice of the tender bride calling alone in her measure that nonsense of patriarchal impiety. It was all the more gratifying when I heard that the Governor of Shetton is now apparently labouring in the same breath, when he speaks"
|
| 57 |
+
],
|
| 58 |
+
"training_time_seconds": 9008.03685593605,
|
| 59 |
+
"stage_training_flops": 1.9399969190612828e+18,
|
| 60 |
+
"inherited_parent_flops": 0.0,
|
| 61 |
+
"cumulative_pipeline_training_flops": 1.9399969190612828e+18,
|
| 62 |
+
"config_fingerprint": "407a5074e0bf3730",
|
| 63 |
+
"git_commit_sha": "083cd7f99484b5e894a23e4a093de6d339c412ea",
|
| 64 |
+
"wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/e3483a4b",
|
| 65 |
+
"huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r11.25-ctx8192",
|
| 66 |
+
"dataset_fingerprint": "63a5e6be81591d82",
|
| 67 |
+
"tokenizer_fingerprint": "03c4f62e7a9d0c3b",
|
| 68 |
+
"unique_train_tokens": 0
|
| 69 |
+
}
|
experiments/think-d12-r11.25-ctx8192/tokenizer/experiment_tokenizer.json
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"experiment_id": "think-d12-r11.25-ctx8192",
|
| 3 |
+
"dataset": {
|
| 4 |
+
"adapter": "parquet_shards",
|
| 5 |
+
"repo": "jbduran/think-dataset",
|
| 6 |
+
"revision": "main",
|
| 7 |
+
"validation_shard": 472,
|
| 8 |
+
"num_train_shards": 24,
|
| 9 |
+
"download_workers": 4
|
| 10 |
+
},
|
| 11 |
+
"tokenizer": {
|
| 12 |
+
"mode": "train",
|
| 13 |
+
"max_chars": 2000000000,
|
| 14 |
+
"doc_cap": 10000,
|
| 15 |
+
"vocab_size": 32768
|
| 16 |
+
},
|
| 17 |
+
"created_at": 1783708207
|
| 18 |
+
}
|
experiments/think-d12-r11.25-ctx8192/tokenizer/token_bytes.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1
|
| 3 |
+
size 132649
|
experiments/think-d12-r11.25-ctx8192/tokenizer/tokenizer.pkl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1
|
| 3 |
+
size 404071
|
experiments/think-d12-r11.25/base_checkpoints/meta_000500.json
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 500,
|
| 3 |
+
"val_bpb": null,
|
| 4 |
+
"model_config": {
|
| 5 |
+
"sequence_len": 2048,
|
| 6 |
+
"vocab_size": 32768,
|
| 7 |
+
"n_layer": 12,
|
| 8 |
+
"n_head": 6,
|
| 9 |
+
"n_kv_head": 6,
|
| 10 |
+
"n_embd": 768,
|
| 11 |
+
"window_pattern": "L"
|
| 12 |
+
},
|
| 13 |
+
"user_config": {
|
| 14 |
+
"run": "dummy",
|
| 15 |
+
"device_type": "",
|
| 16 |
+
"fp8": false,
|
| 17 |
+
"fp8_recipe": "tensorwise",
|
| 18 |
+
"depth": 12,
|
| 19 |
+
"aspect_ratio": 64,
|
| 20 |
+
"head_dim": 128,
|
| 21 |
+
"max_seq_len": 2048,
|
| 22 |
+
"window_pattern": "L",
|
| 23 |
+
"num_iterations": -1,
|
| 24 |
+
"target_flops": -1.0,
|
| 25 |
+
"target_param_data_ratio": 11.25,
|
| 26 |
+
"device_batch_size": 16,
|
| 27 |
+
"total_batch_size": -1,
|
| 28 |
+
"embedding_lr": 0.3,
|
| 29 |
+
"unembedding_lr": 0.008,
|
| 30 |
+
"weight_decay": 0.28,
|
| 31 |
+
"matrix_lr": 0.02,
|
| 32 |
+
"scalar_lr": 0.5,
|
| 33 |
+
"warmup_steps": 40,
|
| 34 |
+
"warmdown_ratio": 0.65,
|
| 35 |
+
"final_lr_frac": 0.05,
|
| 36 |
+
"resume_from_step": -1,
|
| 37 |
+
"pretokenized": true,
|
| 38 |
+
"eval_every": -1,
|
| 39 |
+
"eval_tokens": 41943040,
|
| 40 |
+
"core_metric_every": -1,
|
| 41 |
+
"core_metric_max_per_task": 500,
|
| 42 |
+
"sample_every": -1,
|
| 43 |
+
"save_every": 500,
|
| 44 |
+
"model_tag": null
|
| 45 |
+
},
|
| 46 |
+
"device_batch_size": 16,
|
| 47 |
+
"max_seq_len": 2048,
|
| 48 |
+
"total_batch_size": 524288,
|
| 49 |
+
"dataloader_state_dict": {
|
| 50 |
+
"file_idx": 2,
|
| 51 |
+
"pos": 62184769,
|
| 52 |
+
"epoch": 1,
|
| 53 |
+
"pq_idx": 2,
|
| 54 |
+
"rg_idx": 62184769
|
| 55 |
+
},
|
| 56 |
+
"loop_state": {
|
| 57 |
+
"min_val_bpb": Infinity,
|
| 58 |
+
"smooth_train_loss": 3.5893779623775406,
|
| 59 |
+
"total_training_time": 1290.4532148838043
|
| 60 |
+
}
|
| 61 |
+
}
|
experiments/think-d12-r11.25/base_checkpoints/meta_001000.json
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 1000,
|
| 3 |
+
"val_bpb": null,
|
| 4 |
+
"model_config": {
|
| 5 |
+
"sequence_len": 2048,
|
| 6 |
+
"vocab_size": 32768,
|
| 7 |
+
"n_layer": 12,
|
| 8 |
+
"n_head": 6,
|
| 9 |
+
"n_kv_head": 6,
|
| 10 |
+
"n_embd": 768,
|
| 11 |
+
"window_pattern": "L"
|
| 12 |
+
},
|
| 13 |
+
"user_config": {
|
| 14 |
+
"run": "dummy",
|
| 15 |
+
"device_type": "",
|
| 16 |
+
"fp8": false,
|
| 17 |
+
"fp8_recipe": "tensorwise",
|
| 18 |
+
"depth": 12,
|
| 19 |
+
"aspect_ratio": 64,
|
| 20 |
+
"head_dim": 128,
|
| 21 |
+
"max_seq_len": 2048,
|
| 22 |
+
"window_pattern": "L",
|
| 23 |
+
"num_iterations": -1,
|
| 24 |
+
"target_flops": -1.0,
|
| 25 |
+
"target_param_data_ratio": 11.25,
|
| 26 |
+
"device_batch_size": 16,
|
| 27 |
+
"total_batch_size": -1,
|
| 28 |
+
"embedding_lr": 0.3,
|
| 29 |
+
"unembedding_lr": 0.008,
|
| 30 |
+
"weight_decay": 0.28,
|
| 31 |
+
"matrix_lr": 0.02,
|
| 32 |
+
"scalar_lr": 0.5,
|
| 33 |
+
"warmup_steps": 40,
|
| 34 |
+
"warmdown_ratio": 0.65,
|
| 35 |
+
"final_lr_frac": 0.05,
|
| 36 |
+
"resume_from_step": -1,
|
| 37 |
+
"pretokenized": true,
|
| 38 |
+
"eval_every": -1,
|
| 39 |
+
"eval_tokens": 41943040,
|
| 40 |
+
"core_metric_every": -1,
|
| 41 |
+
"core_metric_max_per_task": 500,
|
| 42 |
+
"sample_every": -1,
|
| 43 |
+
"save_every": 500,
|
| 44 |
+
"model_tag": null
|
| 45 |
+
},
|
| 46 |
+
"device_batch_size": 16,
|
| 47 |
+
"max_seq_len": 2048,
|
| 48 |
+
"total_batch_size": 524288,
|
| 49 |
+
"dataloader_state_dict": {
|
| 50 |
+
"file_idx": 5,
|
| 51 |
+
"pos": 24336769,
|
| 52 |
+
"epoch": 1,
|
| 53 |
+
"pq_idx": 5,
|
| 54 |
+
"rg_idx": 24336769
|
| 55 |
+
},
|
| 56 |
+
"loop_state": {
|
| 57 |
+
"min_val_bpb": Infinity,
|
| 58 |
+
"smooth_train_loss": 3.4419165825253133,
|
| 59 |
+
"total_training_time": 2609.577807664871
|
| 60 |
+
}
|
| 61 |
+
}
|
experiments/think-d12-r11.25/base_checkpoints/meta_001500.json
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 1500,
|
| 3 |
+
"val_bpb": null,
|
| 4 |
+
"model_config": {
|
| 5 |
+
"sequence_len": 2048,
|
| 6 |
+
"vocab_size": 32768,
|
| 7 |
+
"n_layer": 12,
|
| 8 |
+
"n_head": 6,
|
| 9 |
+
"n_kv_head": 6,
|
| 10 |
+
"n_embd": 768,
|
| 11 |
+
"window_pattern": "L"
|
| 12 |
+
},
|
| 13 |
+
"user_config": {
|
| 14 |
+
"run": "dummy",
|
| 15 |
+
"device_type": "",
|
| 16 |
+
"fp8": false,
|
| 17 |
+
"fp8_recipe": "tensorwise",
|
| 18 |
+
"depth": 12,
|
| 19 |
+
"aspect_ratio": 64,
|
| 20 |
+
"head_dim": 128,
|
| 21 |
+
"max_seq_len": 2048,
|
| 22 |
+
"window_pattern": "L",
|
| 23 |
+
"num_iterations": -1,
|
| 24 |
+
"target_flops": -1.0,
|
| 25 |
+
"target_param_data_ratio": 11.25,
|
| 26 |
+
"device_batch_size": 16,
|
| 27 |
+
"total_batch_size": -1,
|
| 28 |
+
"embedding_lr": 0.3,
|
| 29 |
+
"unembedding_lr": 0.008,
|
| 30 |
+
"weight_decay": 0.28,
|
| 31 |
+
"matrix_lr": 0.02,
|
| 32 |
+
"scalar_lr": 0.5,
|
| 33 |
+
"warmup_steps": 40,
|
| 34 |
+
"warmdown_ratio": 0.65,
|
| 35 |
+
"final_lr_frac": 0.05,
|
| 36 |
+
"resume_from_step": -1,
|
| 37 |
+
"pretokenized": true,
|
| 38 |
+
"eval_every": -1,
|
| 39 |
+
"eval_tokens": 41943040,
|
| 40 |
+
"core_metric_every": -1,
|
| 41 |
+
"core_metric_max_per_task": 500,
|
| 42 |
+
"sample_every": -1,
|
| 43 |
+
"save_every": 500,
|
| 44 |
+
"model_tag": null
|
| 45 |
+
},
|
| 46 |
+
"device_batch_size": 16,
|
| 47 |
+
"max_seq_len": 2048,
|
| 48 |
+
"total_batch_size": 524288,
|
| 49 |
+
"dataloader_state_dict": {
|
| 50 |
+
"file_idx": 7,
|
| 51 |
+
"pos": 86488769,
|
| 52 |
+
"epoch": 1,
|
| 53 |
+
"pq_idx": 7,
|
| 54 |
+
"rg_idx": 86488769
|
| 55 |
+
},
|
| 56 |
+
"loop_state": {
|
| 57 |
+
"min_val_bpb": Infinity,
|
| 58 |
+
"smooth_train_loss": 3.2465426140603015,
|
| 59 |
+
"total_training_time": 3929.598204135895
|
| 60 |
+
}
|
| 61 |
+
}
|
experiments/think-d12-r11.25/base_checkpoints/meta_002000.json
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 2000,
|
| 3 |
+
"val_bpb": null,
|
| 4 |
+
"model_config": {
|
| 5 |
+
"sequence_len": 2048,
|
| 6 |
+
"vocab_size": 32768,
|
| 7 |
+
"n_layer": 12,
|
| 8 |
+
"n_head": 6,
|
| 9 |
+
"n_kv_head": 6,
|
| 10 |
+
"n_embd": 768,
|
| 11 |
+
"window_pattern": "L"
|
| 12 |
+
},
|
| 13 |
+
"user_config": {
|
| 14 |
+
"run": "dummy",
|
| 15 |
+
"device_type": "",
|
| 16 |
+
"fp8": false,
|
| 17 |
+
"fp8_recipe": "tensorwise",
|
| 18 |
+
"depth": 12,
|
| 19 |
+
"aspect_ratio": 64,
|
| 20 |
+
"head_dim": 128,
|
| 21 |
+
"max_seq_len": 2048,
|
| 22 |
+
"window_pattern": "L",
|
| 23 |
+
"num_iterations": -1,
|
| 24 |
+
"target_flops": -1.0,
|
| 25 |
+
"target_param_data_ratio": 11.25,
|
| 26 |
+
"device_batch_size": 16,
|
| 27 |
+
"total_batch_size": -1,
|
| 28 |
+
"embedding_lr": 0.3,
|
| 29 |
+
"unembedding_lr": 0.008,
|
| 30 |
+
"weight_decay": 0.28,
|
| 31 |
+
"matrix_lr": 0.02,
|
| 32 |
+
"scalar_lr": 0.5,
|
| 33 |
+
"warmup_steps": 40,
|
| 34 |
+
"warmdown_ratio": 0.65,
|
| 35 |
+
"final_lr_frac": 0.05,
|
| 36 |
+
"resume_from_step": -1,
|
| 37 |
+
"pretokenized": true,
|
| 38 |
+
"eval_every": -1,
|
| 39 |
+
"eval_tokens": 41943040,
|
| 40 |
+
"core_metric_every": -1,
|
| 41 |
+
"core_metric_max_per_task": 500,
|
| 42 |
+
"sample_every": -1,
|
| 43 |
+
"save_every": 500,
|
| 44 |
+
"model_tag": null
|
| 45 |
+
},
|
| 46 |
+
"device_batch_size": 16,
|
| 47 |
+
"max_seq_len": 2048,
|
| 48 |
+
"total_batch_size": 524288,
|
| 49 |
+
"dataloader_state_dict": {
|
| 50 |
+
"file_idx": 10,
|
| 51 |
+
"pos": 48640769,
|
| 52 |
+
"epoch": 1,
|
| 53 |
+
"pq_idx": 10,
|
| 54 |
+
"rg_idx": 48640769
|
| 55 |
+
},
|
| 56 |
+
"loop_state": {
|
| 57 |
+
"min_val_bpb": Infinity,
|
| 58 |
+
"smooth_train_loss": 3.2442641345309373,
|
| 59 |
+
"total_training_time": 5249.494728565216
|
| 60 |
+
}
|
| 61 |
+
}
|
experiments/think-d12-r11.25/base_checkpoints/meta_002362.json
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 2362,
|
| 3 |
+
"val_bpb": null,
|
| 4 |
+
"model_config": {
|
| 5 |
+
"sequence_len": 2048,
|
| 6 |
+
"vocab_size": 32768,
|
| 7 |
+
"n_layer": 12,
|
| 8 |
+
"n_head": 6,
|
| 9 |
+
"n_kv_head": 6,
|
| 10 |
+
"n_embd": 768,
|
| 11 |
+
"window_pattern": "L"
|
| 12 |
+
},
|
| 13 |
+
"user_config": {
|
| 14 |
+
"run": "dummy",
|
| 15 |
+
"device_type": "",
|
| 16 |
+
"fp8": false,
|
| 17 |
+
"fp8_recipe": "tensorwise",
|
| 18 |
+
"depth": 12,
|
| 19 |
+
"aspect_ratio": 64,
|
| 20 |
+
"head_dim": 128,
|
| 21 |
+
"max_seq_len": 2048,
|
| 22 |
+
"window_pattern": "L",
|
| 23 |
+
"num_iterations": -1,
|
| 24 |
+
"target_flops": -1.0,
|
| 25 |
+
"target_param_data_ratio": 11.25,
|
| 26 |
+
"device_batch_size": 16,
|
| 27 |
+
"total_batch_size": -1,
|
| 28 |
+
"embedding_lr": 0.3,
|
| 29 |
+
"unembedding_lr": 0.008,
|
| 30 |
+
"weight_decay": 0.28,
|
| 31 |
+
"matrix_lr": 0.02,
|
| 32 |
+
"scalar_lr": 0.5,
|
| 33 |
+
"warmup_steps": 40,
|
| 34 |
+
"warmdown_ratio": 0.65,
|
| 35 |
+
"final_lr_frac": 0.05,
|
| 36 |
+
"resume_from_step": -1,
|
| 37 |
+
"pretokenized": true,
|
| 38 |
+
"eval_every": -1,
|
| 39 |
+
"eval_tokens": 41943040,
|
| 40 |
+
"core_metric_every": -1,
|
| 41 |
+
"core_metric_max_per_task": 500,
|
| 42 |
+
"sample_every": -1,
|
| 43 |
+
"save_every": 500,
|
| 44 |
+
"model_tag": null
|
| 45 |
+
},
|
| 46 |
+
"device_batch_size": 16,
|
| 47 |
+
"max_seq_len": 2048,
|
| 48 |
+
"total_batch_size": 524288,
|
| 49 |
+
"dataloader_state_dict": {
|
| 50 |
+
"file_idx": 12,
|
| 51 |
+
"pos": 38438817,
|
| 52 |
+
"epoch": 1,
|
| 53 |
+
"pq_idx": 12,
|
| 54 |
+
"rg_idx": 38438817
|
| 55 |
+
},
|
| 56 |
+
"loop_state": {
|
| 57 |
+
"min_val_bpb": Infinity,
|
| 58 |
+
"smooth_train_loss": 3.074421420856799,
|
| 59 |
+
"total_training_time": 6205.646646976471
|
| 60 |
+
}
|
| 61 |
+
}
|
experiments/think-d12-r11.25/base_checkpoints/model_000500.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:21fbdff366e5db3fa2f254b4b98c763cc70ba95722242fa032d8d21b956b694f
|
| 3 |
+
size 792761399
|
experiments/think-d12-r11.25/base_checkpoints/model_001000.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6f0a862044e51b8cac250cf1e4f26af7f8347fd887b8c30af08addb879b6494d
|
| 3 |
+
size 792761399
|
experiments/think-d12-r11.25/base_checkpoints/model_001500.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:326bddc26c100de33310f9164dc873af6099a9b7760527a8707c256988a8ec7a
|
| 3 |
+
size 792761399
|
experiments/think-d12-r11.25/base_checkpoints/model_002000.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a5d101975722cffc43eb4f35aa967a5afd397b159da3521e5a1acbd3819d466e
|
| 3 |
+
size 792761399
|
experiments/think-d12-r11.25/base_checkpoints/model_002362.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:04331c52ed8fa7259f350e4ec72d0dd6602451cfd75a5773a4c17ac5c141ea7e
|
| 3 |
+
size 792761399
|
experiments/think-d12-r11.25/base_checkpoints/optim_000500_rank0.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d39a131089a476a202ae932f21d9398b225d0b8e4147a58fbdff797914d34976
|
| 3 |
+
size 1246165237
|
experiments/think-d12-r11.25/base_checkpoints/optim_001000_rank0.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e4971566eb3d4aa5d5fe29b3a3b77f56581b61713e5c1d162debb8a409c02118
|
| 3 |
+
size 1246165237
|
experiments/think-d12-r11.25/base_checkpoints/optim_001500_rank0.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:05ce87d44bfd6e9b4d4c1e643a2a5aa4b169119913ccbdf3625d7cfa7313b342
|
| 3 |
+
size 1246165237
|
experiments/think-d12-r11.25/base_checkpoints/optim_002000_rank0.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1b58f68fd887360ce99e8456aecf706b1bcf5c8210194721389698ec658237b1
|
| 3 |
+
size 1246165237
|
experiments/think-d12-r11.25/base_checkpoints/optim_002362_rank0.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:09aa5632a4b33981a1b2fbd97254d0050ecb7916d0c242f1d195c0726be314d7
|
| 3 |
+
size 1246165237
|
experiments/think-d12-r11.25/config.json
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"schema_version": 1,
|
| 3 |
+
"stage": "base",
|
| 4 |
+
"experiment_id": "think-d12-r11.25",
|
| 5 |
+
"dataset": {
|
| 6 |
+
"adapter": "parquet_shards",
|
| 7 |
+
"repo": "jbduran/think-dataset",
|
| 8 |
+
"revision": "main",
|
| 9 |
+
"validation_shard": 472,
|
| 10 |
+
"num_train_shards": 24,
|
| 11 |
+
"download_workers": 4
|
| 12 |
+
},
|
| 13 |
+
"tokenizer": {
|
| 14 |
+
"mode": "train",
|
| 15 |
+
"max_chars": 2000000000,
|
| 16 |
+
"doc_cap": 10000,
|
| 17 |
+
"vocab_size": 32768
|
| 18 |
+
},
|
| 19 |
+
"pretokenize": {
|
| 20 |
+
"enabled": true,
|
| 21 |
+
"slack": 1.03,
|
| 22 |
+
"val_tokens": 20971520,
|
| 23 |
+
"shard_tokens": 100000000,
|
| 24 |
+
"tokenizer_threads": 8
|
| 25 |
+
},
|
| 26 |
+
"training": {
|
| 27 |
+
"depth": 12,
|
| 28 |
+
"scaling_params": 110100912,
|
| 29 |
+
"target_param_data_ratio": 11.25,
|
| 30 |
+
"window_pattern": "L",
|
| 31 |
+
"device_batch_size": 16,
|
| 32 |
+
"total_batch_size": 524288,
|
| 33 |
+
"save_every": 500,
|
| 34 |
+
"eval_every": 250,
|
| 35 |
+
"eval_tokens": 2097152,
|
| 36 |
+
"core_metric_every": -1,
|
| 37 |
+
"sample_every": -1
|
| 38 |
+
},
|
| 39 |
+
"artifacts": {
|
| 40 |
+
"repo": "jbduran/think.nano"
|
| 41 |
+
},
|
| 42 |
+
"wandb": {
|
| 43 |
+
"entity": "jbduran-thinkingmachinesncsu",
|
| 44 |
+
"project": "think.nano",
|
| 45 |
+
"name": "think-d12-r11.25",
|
| 46 |
+
"group": "think-d12",
|
| 47 |
+
"tags": [
|
| 48 |
+
"think-dataset",
|
| 49 |
+
"d12",
|
| 50 |
+
"ratio11.25"
|
| 51 |
+
]
|
| 52 |
+
},
|
| 53 |
+
"config_fingerprint": "3b5a68714770b6af",
|
| 54 |
+
"artifact_path": "experiments/think-d12-r11.25"
|
| 55 |
+
}
|
experiments/think-d12-r11.25/evals/samples.json
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "base_model (step 2362)",
|
| 3 |
+
"step": 2362,
|
| 4 |
+
"bpb": {},
|
| 5 |
+
"core_metric": null,
|
| 6 |
+
"core_results": null,
|
| 7 |
+
"centered_results": null,
|
| 8 |
+
"conditioned_samples": [
|
| 9 |
+
{
|
| 10 |
+
"prompt": "The capital of France is",
|
| 11 |
+
"text": "<|bos|>The capital of France is 10,000,000 francs, and the capital of the United"
|
| 12 |
+
},
|
| 13 |
+
{
|
| 14 |
+
"prompt": "The chemical symbol of gold is",
|
| 15 |
+
"text": "<|bos|>The chemical symbol of gold is the gold of the \n\nUnited States. It is the gold of the United States"
|
| 16 |
+
},
|
| 17 |
+
{
|
| 18 |
+
"prompt": "If yesterday was Friday, then tomorrow will be",
|
| 19 |
+
"text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nThe day is Sunday, and the day is Sunday. \n\nThe"
|
| 20 |
+
},
|
| 21 |
+
{
|
| 22 |
+
"prompt": "The opposite of hot is",
|
| 23 |
+
"text": "<|bos|>The opposite of hot is the opposite of cold. \n\nThe opposite of cold is the opposite of cold."
|
| 24 |
+
},
|
| 25 |
+
{
|
| 26 |
+
"prompt": "The planets of the solar system are:",
|
| 27 |
+
"text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, and the stars. \n\n2."
|
| 28 |
+
},
|
| 29 |
+
{
|
| 30 |
+
"prompt": "My favorite color is",
|
| 31 |
+
"text": "<|bos|>My favorite color is the color of the sky. \n\nThe color of the sky is a color of"
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"prompt": "If 5*x + 3 = 13, then x is",
|
| 35 |
+
"text": "<|bos|>If 5*x + 3 = 13, then x is the number of times the number of times the number of times the number of times"
|
| 36 |
+
}
|
| 37 |
+
],
|
| 38 |
+
"unconditioned_samples": [
|
| 39 |
+
"<|bos|>Monthillahivan-this is worthy, they say, of the glory of God's presence, our Father being now to come again! \n\nDead. The priest who said Lord! so between thy guilty hands exulting and curse shoot; do thy honors sweetly, O Father! touch me not with panic, nor miss me with holy surprise, nor sigh (say) tost away my days, nor desolate them; thou wert sent by heaven, and seen with so much care, that Thou near thy sins didst repair them; how art Thou now to come thus early, and bear unto me so divinely? not",
|
| 40 |
+
"<|bos|>370 \n\nPaxton's Introduction to Education, or the true philosophy of the schools criticised, is differentiated. In Rossetti's plan, as already noted, it formulated in these brief articles, there should be but two different versions of the 73\n\n-flat elements; in Rossetti's, we commonly use the broad current of their general aim. That which has been called a positive and material-is aptly called a negative-was given with corresponding emphasis at three different times, at different epochs, but the THREE are frequently gods regular in origin, and preserved from the violence of adjustment and shift, rather than dormant and deliberate meaning of",
|
| 41 |
+
"<|bos|>ROrepoys and Willieot's book on the Siamese Law. \n\nA Letter from Frank Perrivsky to a Prince of Wales. By Kinsman Keith. \n\nTHE \n\nMAY, 1923. \n\nIn Paper \n\nWith 25 Illustrations. Parts. $2 $9 $14 6 $2.50 \n\nIN POLITICAL IDOLSES. \n\nAmerican Law. By George W. Resting, LL.D., Professor of Political Economy in Princeton University. \n\n$18 $2.50 \n\nSamson Minot, hotel-keeper. $3 $7\n\nTHE SCALE OF SIZE.",
|
| 42 |
+
"<|bos|>HENRY MARTYN SAXON, THE HINDUS CHRISTURIENT. \n\nEDWARD IRVING, of the College of the Anatomy School of \n\nDurham, Surrey.\n\nEDWARD PERCY BAKER, OF ALICE COLLEGE, whose personal appearance is now in print, was born at Stanneley, Surrey, July 18, 1794. He has been student in the Company's Military College at Woolwich for more than one year, for his learning and industry in his profession and studies. He has written a book entitled History, Economics and Political Science, which lies nearly at our very door, entitled History, Political Science and Political \n\nScience. It is",
|
| 43 |
+
"<|bos|> HOUSE OF THE ANGELS. \n\nFrom the City of St. Ann. \n\n2 vols. 3s. My Last in a Garden. I reserve for the fifth edition, in manuscript, a full account of ancient His tory. In \n\n1 vol. 5, a short history of Nero and Herod. from 6 to \n\n10 Years, both kept in this Library.\n\n4 \n\nLibrary of the Inducci EN QUANDRON. \n\nFRONTIER, SAMUEL, Dean of Carlisle, President of the\n\nAmerican Board of Works, 3 vols. 2 vols. 3s. 6d. \n\nLibrary",
|
| 44 |
+
"<|bos|>The Bird reflects the world, and swims the eagle's web.-Van Isle.\n\nSHOP'S \n\nREACH \n\nNEGARYELY GENIAL SHOP FINANCIERS RAIDER \n\nREAR HEADS,.} \n\nREPUBLICS, WHILE SHE UNDER FULLER'S PORT, \n\nREP Philosophers, that they be not \n\nPharaoh's patterns good for nothing, and valiant men for that which is nothing; \n\nCLY VAUS\u00d2 EXOVENT \u03b4\u1f72 \u03bf\u03cd\u03c3\u03b1\u03b9 \u0391\u03b4\u03af\u03b4\u03b5\u03b9ANTA\u03c1, \u1f15\u03b4\u03b1\u03c1\u03c9\u03bd \u03c3\u03bf\u03c5\u03b4\u03bf\u1f7a\u03c2 The Queen eschews",
|
| 45 |
+
"<|bos|>'Eau du Monarch beau ou H\u00f4tel de Voodjiches.\" (The same French Inspecteur :) \"Son jours\n\nA \u00e9t\u00e9 autant \u00e0 tomboi les m\u00eames Premi\u00e8res avec une princesse qui sont d'ombres le miraculeur bien de France le souffrir. Les c\u00f4tes de ceux qui le sont j\u00e9sibres.\" (The French Directors.)\n\neffected these river improvements in effeminate cases, in the The Executive has eschewed the corrupdemell\u00e8 hrs. Nieuw Ga",
|
| 46 |
+
"<|bos|>the eight (250) series.\n\nThe pressure is already great.\n\nashions like the palette de la \n\nLast chapter, however, it may be noted, and it is a most common practice in such a period for small and countless series to transform the palette de la up half at the sauce, and it is a matter of considerable importance to seeing the clothes-holders uniformly the marks that indicate the supply of the palette de la into that respectable width whence their sulky tints are generally returned. \n\nThe colour represents the price of the garment at the time apparently labouring in the work. Delivery is possible"
|
| 47 |
+
]
|
| 48 |
+
}
|
experiments/think-d12-r11.25/evals/val_bpb.json
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "base_model (step 2362)",
|
| 3 |
+
"step": 2362,
|
| 4 |
+
"bpb": {
|
| 5 |
+
"val_per_position": [
|
| 6 |
+
{
|
| 7 |
+
"start": 0,
|
| 8 |
+
"end": 256,
|
| 9 |
+
"bpb": 1.1255410237731462
|
| 10 |
+
},
|
| 11 |
+
{
|
| 12 |
+
"start": 256,
|
| 13 |
+
"end": 512,
|
| 14 |
+
"bpb": 1.0636677720059344
|
| 15 |
+
},
|
| 16 |
+
{
|
| 17 |
+
"start": 512,
|
| 18 |
+
"end": 768,
|
| 19 |
+
"bpb": 1.0506832286096848
|
| 20 |
+
},
|
| 21 |
+
{
|
| 22 |
+
"start": 768,
|
| 23 |
+
"end": 1024,
|
| 24 |
+
"bpb": 1.0456638664028364
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"start": 1024,
|
| 28 |
+
"end": 1280,
|
| 29 |
+
"bpb": 1.038324560694976
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"start": 1280,
|
| 33 |
+
"end": 1536,
|
| 34 |
+
"bpb": 1.0330914959872517
|
| 35 |
+
},
|
| 36 |
+
{
|
| 37 |
+
"start": 1536,
|
| 38 |
+
"end": 1792,
|
| 39 |
+
"bpb": 1.0308104068466784
|
| 40 |
+
},
|
| 41 |
+
{
|
| 42 |
+
"start": 1792,
|
| 43 |
+
"end": 2048,
|
| 44 |
+
"bpb": 1.027579795687179
|
| 45 |
+
}
|
| 46 |
+
],
|
| 47 |
+
"val": 1.0519195678472355
|
| 48 |
+
},
|
| 49 |
+
"core_metric": null,
|
| 50 |
+
"core_results": null,
|
| 51 |
+
"centered_results": null,
|
| 52 |
+
"conditioned_samples": [],
|
| 53 |
+
"unconditioned_samples": []
|
| 54 |
+
}
|
experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean-1930s.json
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "base_model (step 2362)",
|
| 3 |
+
"step": 2362,
|
| 4 |
+
"bpb": {
|
| 5 |
+
"val": 1.0781689302026238
|
| 6 |
+
},
|
| 7 |
+
"core_metric": null,
|
| 8 |
+
"core_results": null,
|
| 9 |
+
"centered_results": null,
|
| 10 |
+
"conditioned_samples": [],
|
| 11 |
+
"unconditioned_samples": []
|
| 12 |
+
}
|
experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "base_model (step 2362)",
|
| 3 |
+
"step": 2362,
|
| 4 |
+
"bpb": {
|
| 5 |
+
"val": 1.081385374786877
|
| 6 |
+
},
|
| 7 |
+
"core_metric": null,
|
| 8 |
+
"core_results": null,
|
| 9 |
+
"centered_results": null,
|
| 10 |
+
"conditioned_samples": [],
|
| 11 |
+
"unconditioned_samples": []
|
| 12 |
+
}
|
experiments/think-d12-r11.25/run.json
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"experiment_id": "think-d12-r11.25",
|
| 3 |
+
"stage": "base",
|
| 4 |
+
"wandb_run_id": null,
|
| 5 |
+
"migration_note": "Migrated from the pre-lineage repository layout."
|
| 6 |
+
}
|
experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/meta_001065.json
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"step": 1065,
|
| 3 |
+
"val_bpb": 0.39271438585109175,
|
| 4 |
+
"model_config": {
|
| 5 |
+
"sequence_len": 2048,
|
| 6 |
+
"vocab_size": 32768,
|
| 7 |
+
"n_layer": 12,
|
| 8 |
+
"n_head": 6,
|
| 9 |
+
"n_kv_head": 6,
|
| 10 |
+
"n_embd": 768,
|
| 11 |
+
"window_pattern": "L"
|
| 12 |
+
},
|
| 13 |
+
"user_config": {
|
| 14 |
+
"run": "dummy",
|
| 15 |
+
"device_type": "",
|
| 16 |
+
"model_tag": "d12",
|
| 17 |
+
"model_step": null,
|
| 18 |
+
"load_optimizer": 1,
|
| 19 |
+
"num_iterations": -1,
|
| 20 |
+
"max_seq_len": null,
|
| 21 |
+
"device_batch_size": 8,
|
| 22 |
+
"total_batch_size": null,
|
| 23 |
+
"embedding_lr": null,
|
| 24 |
+
"unembedding_lr": null,
|
| 25 |
+
"matrix_lr": null,
|
| 26 |
+
"init_lr_frac": 0.8,
|
| 27 |
+
"warmup_ratio": 0.0,
|
| 28 |
+
"warmdown_ratio": 0.5,
|
| 29 |
+
"final_lr_frac": 0.0,
|
| 30 |
+
"eval_every": -1,
|
| 31 |
+
"eval_tokens": 20971520,
|
| 32 |
+
"chatcore_every": -1,
|
| 33 |
+
"chatcore_max_cat": -1,
|
| 34 |
+
"chatcore_max_sample": 24,
|
| 35 |
+
"mmlu_epochs": 3,
|
| 36 |
+
"gsm8k_epochs": 4
|
| 37 |
+
}
|
| 38 |
+
}
|
experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/model_001065.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f1ef8886ea2cbd820baa9bc98368f99673efb1f5fdbd082196d573b291111bda
|
| 3 |
+
size 792761399
|
experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/optim_001065_rank0.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:db8c9df6e288618ed6362102e29edc3731267b4ff382154a55709b4d5a14f1d4
|
| 3 |
+
size 1246165237
|