jbduran commited on
Commit
c7dca88
·
verified ·
1 Parent(s): 4b7be98

Import experiment archive from bart (batch 6)

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. experiments/think-d12-r11.25-ctx4096/tokenizer/token_bytes.pt +3 -0
  2. experiments/think-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl +3 -0
  3. experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_000500.json +140 -0
  4. experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001000.json +140 -0
  5. experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001500.json +140 -0
  6. experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002000.json +140 -0
  7. experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002362.json +140 -0
  8. experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_000500.pt +3 -0
  9. experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001000.pt +3 -0
  10. experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001500.pt +3 -0
  11. experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002000.pt +3 -0
  12. experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002362.pt +3 -0
  13. experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_000500_rank0.pt +3 -0
  14. experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001000_rank0.pt +3 -0
  15. experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001500_rank0.pt +3 -0
  16. experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002000_rank0.pt +3 -0
  17. experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002362_rank0.pt +3 -0
  18. experiments/think-d12-r11.25-ctx8192/config.json +57 -0
  19. experiments/think-d12-r11.25-ctx8192/evals/core.json +56 -0
  20. experiments/think-d12-r11.25-ctx8192/evals/samples.json +48 -0
  21. experiments/think-d12-r11.25-ctx8192/evals/val_bpb.json +174 -0
  22. experiments/think-d12-r11.25-ctx8192/run.json +10 -0
  23. experiments/think-d12-r11.25-ctx8192/summary.json +69 -0
  24. experiments/think-d12-r11.25-ctx8192/tokenizer/experiment_tokenizer.json +18 -0
  25. experiments/think-d12-r11.25-ctx8192/tokenizer/token_bytes.pt +3 -0
  26. experiments/think-d12-r11.25-ctx8192/tokenizer/tokenizer.pkl +3 -0
  27. experiments/think-d12-r11.25/base_checkpoints/meta_000500.json +61 -0
  28. experiments/think-d12-r11.25/base_checkpoints/meta_001000.json +61 -0
  29. experiments/think-d12-r11.25/base_checkpoints/meta_001500.json +61 -0
  30. experiments/think-d12-r11.25/base_checkpoints/meta_002000.json +61 -0
  31. experiments/think-d12-r11.25/base_checkpoints/meta_002362.json +61 -0
  32. experiments/think-d12-r11.25/base_checkpoints/model_000500.pt +3 -0
  33. experiments/think-d12-r11.25/base_checkpoints/model_001000.pt +3 -0
  34. experiments/think-d12-r11.25/base_checkpoints/model_001500.pt +3 -0
  35. experiments/think-d12-r11.25/base_checkpoints/model_002000.pt +3 -0
  36. experiments/think-d12-r11.25/base_checkpoints/model_002362.pt +3 -0
  37. experiments/think-d12-r11.25/base_checkpoints/optim_000500_rank0.pt +3 -0
  38. experiments/think-d12-r11.25/base_checkpoints/optim_001000_rank0.pt +3 -0
  39. experiments/think-d12-r11.25/base_checkpoints/optim_001500_rank0.pt +3 -0
  40. experiments/think-d12-r11.25/base_checkpoints/optim_002000_rank0.pt +3 -0
  41. experiments/think-d12-r11.25/base_checkpoints/optim_002362_rank0.pt +3 -0
  42. experiments/think-d12-r11.25/config.json +55 -0
  43. experiments/think-d12-r11.25/evals/samples.json +48 -0
  44. experiments/think-d12-r11.25/evals/val_bpb.json +54 -0
  45. experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean-1930s.json +12 -0
  46. experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json +12 -0
  47. experiments/think-d12-r11.25/run.json +6 -0
  48. experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/meta_001065.json +38 -0
  49. experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/model_001065.pt +3 -0
  50. experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/optim_001065_rank0.pt +3 -0
experiments/think-d12-r11.25-ctx4096/tokenizer/token_bytes.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1
3
+ size 132649
experiments/think-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1
3
+ size 404071
experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_000500.json ADDED
@@ -0,0 +1,140 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "step": 500,
3
+ "training_complete": false,
4
+ "experiment_id": "think-d12-r11.25-ctx8192",
5
+ "val_bpb": 1.2806006574422366,
6
+ "model_config": {
7
+ "sequence_len": 8192,
8
+ "vocab_size": 32768,
9
+ "n_layer": 12,
10
+ "n_head": 6,
11
+ "n_kv_head": 6,
12
+ "n_embd": 768,
13
+ "window_pattern": "L"
14
+ },
15
+ "user_config": {
16
+ "run": "think-d12-r11.25-ctx8192",
17
+ "wandb_run_id": "e3483a4b",
18
+ "wandb_group": "think-d12",
19
+ "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192",
20
+ "device_type": "",
21
+ "fp8": false,
22
+ "fp8_recipe": "tensorwise",
23
+ "depth": 12,
24
+ "aspect_ratio": 64,
25
+ "head_dim": 128,
26
+ "max_seq_len": 8192,
27
+ "window_pattern": "L",
28
+ "num_iterations": -1,
29
+ "target_flops": -1.0,
30
+ "target_param_data_ratio": 11.25,
31
+ "device_batch_size": 4,
32
+ "total_batch_size": 524288,
33
+ "embedding_lr": 0.3,
34
+ "unembedding_lr": 0.008,
35
+ "weight_decay": 0.28,
36
+ "matrix_lr": 0.02,
37
+ "scalar_lr": 0.5,
38
+ "warmup_steps": 40,
39
+ "warmdown_ratio": 0.65,
40
+ "final_lr_frac": 0.05,
41
+ "resume_from_step": -1,
42
+ "pretokenized": true,
43
+ "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data",
44
+ "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer",
45
+ "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok",
46
+ "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints",
47
+ "experiment_id": "think-d12-r11.25-ctx8192",
48
+ "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json",
49
+ "tokenizer_fingerprint": "03c4f62e7a9d0c3b",
50
+ "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847",
51
+ "seed": 42,
52
+ "eval_every": 250,
53
+ "eval_tokens": 2097152,
54
+ "core_metric_every": -1,
55
+ "core_metric_max_per_task": 500,
56
+ "sample_every": -1,
57
+ "save_every": 500,
58
+ "model_tag": "think-d12-r11.25-ctx8192",
59
+ "experiment": {
60
+ "schema_version": 1,
61
+ "stage": "base",
62
+ "experiment_id": "think-d12-r11.25-ctx8192",
63
+ "dataset": {
64
+ "adapter": "parquet_shards",
65
+ "repo": "jbduran/think-dataset",
66
+ "revision": "main",
67
+ "validation_shard": 472,
68
+ "num_train_shards": 24,
69
+ "download_workers": 4
70
+ },
71
+ "tokenizer": {
72
+ "mode": "train",
73
+ "max_chars": 2000000000,
74
+ "doc_cap": 10000,
75
+ "vocab_size": 32768
76
+ },
77
+ "pretokenize": {
78
+ "enabled": true,
79
+ "slack": 1.03,
80
+ "val_tokens": 20971520,
81
+ "shard_tokens": 100000000,
82
+ "tokenizer_threads": 8
83
+ },
84
+ "training": {
85
+ "depth": 12,
86
+ "scaling_params": 110100912,
87
+ "target_param_data_ratio": 11.25,
88
+ "max_seq_len": 8192,
89
+ "window_pattern": "L",
90
+ "device_batch_size": 4,
91
+ "total_batch_size": 524288,
92
+ "save_every": 500,
93
+ "eval_every": 250,
94
+ "eval_tokens": 2097152,
95
+ "core_metric_every": -1,
96
+ "sample_every": -1
97
+ },
98
+ "artifacts": {
99
+ "repo": "jbduran/think.nano"
100
+ },
101
+ "wandb": {
102
+ "entity": "jbduran-thinkingmachinesncsu",
103
+ "project": "think.nano",
104
+ "name": "think-d12-r11.25-ctx8192",
105
+ "group": "think-d12",
106
+ "tags": [
107
+ "think-dataset",
108
+ "d12",
109
+ "ratio11.25",
110
+ "ctx8192"
111
+ ]
112
+ },
113
+ "config_fingerprint": "407a5074e0bf3730",
114
+ "artifact_path": "experiments/think-d12-r11.25-ctx8192"
115
+ },
116
+ "stage": "base",
117
+ "base_experiment_id": "think-d12-r11.25-ctx8192",
118
+ "parent_experiment_id": null,
119
+ "parent_checkpoint_step": null,
120
+ "config_fingerprint": "407a5074e0bf3730"
121
+ },
122
+ "device_batch_size": 4,
123
+ "max_seq_len": 8192,
124
+ "total_batch_size": 524288,
125
+ "dataloader_state_dict": {
126
+ "file_idx": 2,
127
+ "pos": 62184769,
128
+ "epoch": 1,
129
+ "pq_idx": 2,
130
+ "rg_idx": 62184769
131
+ },
132
+ "loop_state": {
133
+ "min_val_bpb": 1.2806006574422366,
134
+ "smooth_train_loss": 3.5198863114259193,
135
+ "total_training_time": 1859.6971344947815,
136
+ "stage_training_flops": 410668272451584000,
137
+ "inherited_parent_flops": 0.0,
138
+ "cumulative_pipeline_training_flops": 410668272451584000
139
+ }
140
+ }
experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001000.json ADDED
@@ -0,0 +1,140 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "step": 1000,
3
+ "training_complete": false,
4
+ "experiment_id": "think-d12-r11.25-ctx8192",
5
+ "val_bpb": 1.1673074973592756,
6
+ "model_config": {
7
+ "sequence_len": 8192,
8
+ "vocab_size": 32768,
9
+ "n_layer": 12,
10
+ "n_head": 6,
11
+ "n_kv_head": 6,
12
+ "n_embd": 768,
13
+ "window_pattern": "L"
14
+ },
15
+ "user_config": {
16
+ "run": "think-d12-r11.25-ctx8192",
17
+ "wandb_run_id": "e3483a4b",
18
+ "wandb_group": "think-d12",
19
+ "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192",
20
+ "device_type": "",
21
+ "fp8": false,
22
+ "fp8_recipe": "tensorwise",
23
+ "depth": 12,
24
+ "aspect_ratio": 64,
25
+ "head_dim": 128,
26
+ "max_seq_len": 8192,
27
+ "window_pattern": "L",
28
+ "num_iterations": -1,
29
+ "target_flops": -1.0,
30
+ "target_param_data_ratio": 11.25,
31
+ "device_batch_size": 4,
32
+ "total_batch_size": 524288,
33
+ "embedding_lr": 0.3,
34
+ "unembedding_lr": 0.008,
35
+ "weight_decay": 0.28,
36
+ "matrix_lr": 0.02,
37
+ "scalar_lr": 0.5,
38
+ "warmup_steps": 40,
39
+ "warmdown_ratio": 0.65,
40
+ "final_lr_frac": 0.05,
41
+ "resume_from_step": -1,
42
+ "pretokenized": true,
43
+ "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data",
44
+ "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer",
45
+ "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok",
46
+ "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints",
47
+ "experiment_id": "think-d12-r11.25-ctx8192",
48
+ "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json",
49
+ "tokenizer_fingerprint": "03c4f62e7a9d0c3b",
50
+ "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847",
51
+ "seed": 42,
52
+ "eval_every": 250,
53
+ "eval_tokens": 2097152,
54
+ "core_metric_every": -1,
55
+ "core_metric_max_per_task": 500,
56
+ "sample_every": -1,
57
+ "save_every": 500,
58
+ "model_tag": "think-d12-r11.25-ctx8192",
59
+ "experiment": {
60
+ "schema_version": 1,
61
+ "stage": "base",
62
+ "experiment_id": "think-d12-r11.25-ctx8192",
63
+ "dataset": {
64
+ "adapter": "parquet_shards",
65
+ "repo": "jbduran/think-dataset",
66
+ "revision": "main",
67
+ "validation_shard": 472,
68
+ "num_train_shards": 24,
69
+ "download_workers": 4
70
+ },
71
+ "tokenizer": {
72
+ "mode": "train",
73
+ "max_chars": 2000000000,
74
+ "doc_cap": 10000,
75
+ "vocab_size": 32768
76
+ },
77
+ "pretokenize": {
78
+ "enabled": true,
79
+ "slack": 1.03,
80
+ "val_tokens": 20971520,
81
+ "shard_tokens": 100000000,
82
+ "tokenizer_threads": 8
83
+ },
84
+ "training": {
85
+ "depth": 12,
86
+ "scaling_params": 110100912,
87
+ "target_param_data_ratio": 11.25,
88
+ "max_seq_len": 8192,
89
+ "window_pattern": "L",
90
+ "device_batch_size": 4,
91
+ "total_batch_size": 524288,
92
+ "save_every": 500,
93
+ "eval_every": 250,
94
+ "eval_tokens": 2097152,
95
+ "core_metric_every": -1,
96
+ "sample_every": -1
97
+ },
98
+ "artifacts": {
99
+ "repo": "jbduran/think.nano"
100
+ },
101
+ "wandb": {
102
+ "entity": "jbduran-thinkingmachinesncsu",
103
+ "project": "think.nano",
104
+ "name": "think-d12-r11.25-ctx8192",
105
+ "group": "think-d12",
106
+ "tags": [
107
+ "think-dataset",
108
+ "d12",
109
+ "ratio11.25",
110
+ "ctx8192"
111
+ ]
112
+ },
113
+ "config_fingerprint": "407a5074e0bf3730",
114
+ "artifact_path": "experiments/think-d12-r11.25-ctx8192"
115
+ },
116
+ "stage": "base",
117
+ "base_experiment_id": "think-d12-r11.25-ctx8192",
118
+ "parent_experiment_id": null,
119
+ "parent_checkpoint_step": null,
120
+ "config_fingerprint": "407a5074e0bf3730"
121
+ },
122
+ "device_batch_size": 4,
123
+ "max_seq_len": 8192,
124
+ "total_batch_size": 524288,
125
+ "dataloader_state_dict": {
126
+ "file_idx": 5,
127
+ "pos": 24336769,
128
+ "epoch": 1,
129
+ "pq_idx": 5,
130
+ "rg_idx": 24336769
131
+ },
132
+ "loop_state": {
133
+ "min_val_bpb": 1.1673074973592756,
134
+ "smooth_train_loss": 3.291130702382907,
135
+ "total_training_time": 3762.280524253845,
136
+ "stage_training_flops": 821336544903168000,
137
+ "inherited_parent_flops": 0.0,
138
+ "cumulative_pipeline_training_flops": 821336544903168000
139
+ }
140
+ }
experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001500.json ADDED
@@ -0,0 +1,140 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "step": 1500,
3
+ "training_complete": false,
4
+ "experiment_id": "think-d12-r11.25-ctx8192",
5
+ "val_bpb": 1.108596107059193,
6
+ "model_config": {
7
+ "sequence_len": 8192,
8
+ "vocab_size": 32768,
9
+ "n_layer": 12,
10
+ "n_head": 6,
11
+ "n_kv_head": 6,
12
+ "n_embd": 768,
13
+ "window_pattern": "L"
14
+ },
15
+ "user_config": {
16
+ "run": "think-d12-r11.25-ctx8192",
17
+ "wandb_run_id": "e3483a4b",
18
+ "wandb_group": "think-d12",
19
+ "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192",
20
+ "device_type": "",
21
+ "fp8": false,
22
+ "fp8_recipe": "tensorwise",
23
+ "depth": 12,
24
+ "aspect_ratio": 64,
25
+ "head_dim": 128,
26
+ "max_seq_len": 8192,
27
+ "window_pattern": "L",
28
+ "num_iterations": -1,
29
+ "target_flops": -1.0,
30
+ "target_param_data_ratio": 11.25,
31
+ "device_batch_size": 4,
32
+ "total_batch_size": 524288,
33
+ "embedding_lr": 0.3,
34
+ "unembedding_lr": 0.008,
35
+ "weight_decay": 0.28,
36
+ "matrix_lr": 0.02,
37
+ "scalar_lr": 0.5,
38
+ "warmup_steps": 40,
39
+ "warmdown_ratio": 0.65,
40
+ "final_lr_frac": 0.05,
41
+ "resume_from_step": -1,
42
+ "pretokenized": true,
43
+ "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data",
44
+ "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer",
45
+ "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok",
46
+ "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints",
47
+ "experiment_id": "think-d12-r11.25-ctx8192",
48
+ "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json",
49
+ "tokenizer_fingerprint": "03c4f62e7a9d0c3b",
50
+ "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847",
51
+ "seed": 42,
52
+ "eval_every": 250,
53
+ "eval_tokens": 2097152,
54
+ "core_metric_every": -1,
55
+ "core_metric_max_per_task": 500,
56
+ "sample_every": -1,
57
+ "save_every": 500,
58
+ "model_tag": "think-d12-r11.25-ctx8192",
59
+ "experiment": {
60
+ "schema_version": 1,
61
+ "stage": "base",
62
+ "experiment_id": "think-d12-r11.25-ctx8192",
63
+ "dataset": {
64
+ "adapter": "parquet_shards",
65
+ "repo": "jbduran/think-dataset",
66
+ "revision": "main",
67
+ "validation_shard": 472,
68
+ "num_train_shards": 24,
69
+ "download_workers": 4
70
+ },
71
+ "tokenizer": {
72
+ "mode": "train",
73
+ "max_chars": 2000000000,
74
+ "doc_cap": 10000,
75
+ "vocab_size": 32768
76
+ },
77
+ "pretokenize": {
78
+ "enabled": true,
79
+ "slack": 1.03,
80
+ "val_tokens": 20971520,
81
+ "shard_tokens": 100000000,
82
+ "tokenizer_threads": 8
83
+ },
84
+ "training": {
85
+ "depth": 12,
86
+ "scaling_params": 110100912,
87
+ "target_param_data_ratio": 11.25,
88
+ "max_seq_len": 8192,
89
+ "window_pattern": "L",
90
+ "device_batch_size": 4,
91
+ "total_batch_size": 524288,
92
+ "save_every": 500,
93
+ "eval_every": 250,
94
+ "eval_tokens": 2097152,
95
+ "core_metric_every": -1,
96
+ "sample_every": -1
97
+ },
98
+ "artifacts": {
99
+ "repo": "jbduran/think.nano"
100
+ },
101
+ "wandb": {
102
+ "entity": "jbduran-thinkingmachinesncsu",
103
+ "project": "think.nano",
104
+ "name": "think-d12-r11.25-ctx8192",
105
+ "group": "think-d12",
106
+ "tags": [
107
+ "think-dataset",
108
+ "d12",
109
+ "ratio11.25",
110
+ "ctx8192"
111
+ ]
112
+ },
113
+ "config_fingerprint": "407a5074e0bf3730",
114
+ "artifact_path": "experiments/think-d12-r11.25-ctx8192"
115
+ },
116
+ "stage": "base",
117
+ "base_experiment_id": "think-d12-r11.25-ctx8192",
118
+ "parent_experiment_id": null,
119
+ "parent_checkpoint_step": null,
120
+ "config_fingerprint": "407a5074e0bf3730"
121
+ },
122
+ "device_batch_size": 4,
123
+ "max_seq_len": 8192,
124
+ "total_batch_size": 524288,
125
+ "dataloader_state_dict": {
126
+ "file_idx": 7,
127
+ "pos": 86488769,
128
+ "epoch": 1,
129
+ "pq_idx": 7,
130
+ "rg_idx": 86488769
131
+ },
132
+ "loop_state": {
133
+ "min_val_bpb": 1.108596107059193,
134
+ "smooth_train_loss": 3.1279163553930784,
135
+ "total_training_time": 5663.0339615345,
136
+ "stage_training_flops": 1232004817354752000,
137
+ "inherited_parent_flops": 0.0,
138
+ "cumulative_pipeline_training_flops": 1232004817354752000
139
+ }
140
+ }
experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002000.json ADDED
@@ -0,0 +1,140 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "step": 2000,
3
+ "training_complete": false,
4
+ "experiment_id": "think-d12-r11.25-ctx8192",
5
+ "val_bpb": 1.0589838913914653,
6
+ "model_config": {
7
+ "sequence_len": 8192,
8
+ "vocab_size": 32768,
9
+ "n_layer": 12,
10
+ "n_head": 6,
11
+ "n_kv_head": 6,
12
+ "n_embd": 768,
13
+ "window_pattern": "L"
14
+ },
15
+ "user_config": {
16
+ "run": "think-d12-r11.25-ctx8192",
17
+ "wandb_run_id": "e3483a4b",
18
+ "wandb_group": "think-d12",
19
+ "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192",
20
+ "device_type": "",
21
+ "fp8": false,
22
+ "fp8_recipe": "tensorwise",
23
+ "depth": 12,
24
+ "aspect_ratio": 64,
25
+ "head_dim": 128,
26
+ "max_seq_len": 8192,
27
+ "window_pattern": "L",
28
+ "num_iterations": -1,
29
+ "target_flops": -1.0,
30
+ "target_param_data_ratio": 11.25,
31
+ "device_batch_size": 4,
32
+ "total_batch_size": 524288,
33
+ "embedding_lr": 0.3,
34
+ "unembedding_lr": 0.008,
35
+ "weight_decay": 0.28,
36
+ "matrix_lr": 0.02,
37
+ "scalar_lr": 0.5,
38
+ "warmup_steps": 40,
39
+ "warmdown_ratio": 0.65,
40
+ "final_lr_frac": 0.05,
41
+ "resume_from_step": -1,
42
+ "pretokenized": true,
43
+ "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data",
44
+ "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer",
45
+ "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok",
46
+ "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints",
47
+ "experiment_id": "think-d12-r11.25-ctx8192",
48
+ "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json",
49
+ "tokenizer_fingerprint": "03c4f62e7a9d0c3b",
50
+ "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847",
51
+ "seed": 42,
52
+ "eval_every": 250,
53
+ "eval_tokens": 2097152,
54
+ "core_metric_every": -1,
55
+ "core_metric_max_per_task": 500,
56
+ "sample_every": -1,
57
+ "save_every": 500,
58
+ "model_tag": "think-d12-r11.25-ctx8192",
59
+ "experiment": {
60
+ "schema_version": 1,
61
+ "stage": "base",
62
+ "experiment_id": "think-d12-r11.25-ctx8192",
63
+ "dataset": {
64
+ "adapter": "parquet_shards",
65
+ "repo": "jbduran/think-dataset",
66
+ "revision": "main",
67
+ "validation_shard": 472,
68
+ "num_train_shards": 24,
69
+ "download_workers": 4
70
+ },
71
+ "tokenizer": {
72
+ "mode": "train",
73
+ "max_chars": 2000000000,
74
+ "doc_cap": 10000,
75
+ "vocab_size": 32768
76
+ },
77
+ "pretokenize": {
78
+ "enabled": true,
79
+ "slack": 1.03,
80
+ "val_tokens": 20971520,
81
+ "shard_tokens": 100000000,
82
+ "tokenizer_threads": 8
83
+ },
84
+ "training": {
85
+ "depth": 12,
86
+ "scaling_params": 110100912,
87
+ "target_param_data_ratio": 11.25,
88
+ "max_seq_len": 8192,
89
+ "window_pattern": "L",
90
+ "device_batch_size": 4,
91
+ "total_batch_size": 524288,
92
+ "save_every": 500,
93
+ "eval_every": 250,
94
+ "eval_tokens": 2097152,
95
+ "core_metric_every": -1,
96
+ "sample_every": -1
97
+ },
98
+ "artifacts": {
99
+ "repo": "jbduran/think.nano"
100
+ },
101
+ "wandb": {
102
+ "entity": "jbduran-thinkingmachinesncsu",
103
+ "project": "think.nano",
104
+ "name": "think-d12-r11.25-ctx8192",
105
+ "group": "think-d12",
106
+ "tags": [
107
+ "think-dataset",
108
+ "d12",
109
+ "ratio11.25",
110
+ "ctx8192"
111
+ ]
112
+ },
113
+ "config_fingerprint": "407a5074e0bf3730",
114
+ "artifact_path": "experiments/think-d12-r11.25-ctx8192"
115
+ },
116
+ "stage": "base",
117
+ "base_experiment_id": "think-d12-r11.25-ctx8192",
118
+ "parent_experiment_id": null,
119
+ "parent_checkpoint_step": null,
120
+ "config_fingerprint": "407a5074e0bf3730"
121
+ },
122
+ "device_batch_size": 4,
123
+ "max_seq_len": 8192,
124
+ "total_batch_size": 524288,
125
+ "dataloader_state_dict": {
126
+ "file_idx": 10,
127
+ "pos": 48640769,
128
+ "epoch": 1,
129
+ "pq_idx": 10,
130
+ "rg_idx": 48640769
131
+ },
132
+ "loop_state": {
133
+ "min_val_bpb": 1.0589838913914653,
134
+ "smooth_train_loss": 3.116437961262795,
135
+ "total_training_time": 7566.1544008255005,
136
+ "stage_training_flops": 1642673089806336000,
137
+ "inherited_parent_flops": 0.0,
138
+ "cumulative_pipeline_training_flops": 1642673089806336000
139
+ }
140
+ }
experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002362.json ADDED
@@ -0,0 +1,140 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "step": 2362,
3
+ "training_complete": true,
4
+ "experiment_id": "think-d12-r11.25-ctx8192",
5
+ "val_bpb": 1.0395519592251896,
6
+ "model_config": {
7
+ "sequence_len": 8192,
8
+ "vocab_size": 32768,
9
+ "n_layer": 12,
10
+ "n_head": 6,
11
+ "n_kv_head": 6,
12
+ "n_embd": 768,
13
+ "window_pattern": "L"
14
+ },
15
+ "user_config": {
16
+ "run": "think-d12-r11.25-ctx8192",
17
+ "wandb_run_id": "e3483a4b",
18
+ "wandb_group": "think-d12",
19
+ "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192",
20
+ "device_type": "",
21
+ "fp8": false,
22
+ "fp8_recipe": "tensorwise",
23
+ "depth": 12,
24
+ "aspect_ratio": 64,
25
+ "head_dim": 128,
26
+ "max_seq_len": 8192,
27
+ "window_pattern": "L",
28
+ "num_iterations": -1,
29
+ "target_flops": -1.0,
30
+ "target_param_data_ratio": 11.25,
31
+ "device_batch_size": 4,
32
+ "total_batch_size": 524288,
33
+ "embedding_lr": 0.3,
34
+ "unembedding_lr": 0.008,
35
+ "weight_decay": 0.28,
36
+ "matrix_lr": 0.02,
37
+ "scalar_lr": 0.5,
38
+ "warmup_steps": 40,
39
+ "warmdown_ratio": 0.65,
40
+ "final_lr_frac": 0.05,
41
+ "resume_from_step": 2000,
42
+ "pretokenized": true,
43
+ "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data",
44
+ "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer",
45
+ "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok",
46
+ "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints",
47
+ "experiment_id": "think-d12-r11.25-ctx8192",
48
+ "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json",
49
+ "tokenizer_fingerprint": "03c4f62e7a9d0c3b",
50
+ "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847",
51
+ "seed": 42,
52
+ "eval_every": 250,
53
+ "eval_tokens": 2097152,
54
+ "core_metric_every": -1,
55
+ "core_metric_max_per_task": 500,
56
+ "sample_every": -1,
57
+ "save_every": 500,
58
+ "model_tag": "think-d12-r11.25-ctx8192",
59
+ "experiment": {
60
+ "schema_version": 1,
61
+ "stage": "base",
62
+ "experiment_id": "think-d12-r11.25-ctx8192",
63
+ "dataset": {
64
+ "adapter": "parquet_shards",
65
+ "repo": "jbduran/think-dataset",
66
+ "revision": "main",
67
+ "validation_shard": 472,
68
+ "num_train_shards": 24,
69
+ "download_workers": 4
70
+ },
71
+ "tokenizer": {
72
+ "mode": "train",
73
+ "max_chars": 2000000000,
74
+ "doc_cap": 10000,
75
+ "vocab_size": 32768
76
+ },
77
+ "pretokenize": {
78
+ "enabled": true,
79
+ "slack": 1.03,
80
+ "val_tokens": 20971520,
81
+ "shard_tokens": 100000000,
82
+ "tokenizer_threads": 8
83
+ },
84
+ "training": {
85
+ "depth": 12,
86
+ "scaling_params": 110100912,
87
+ "target_param_data_ratio": 11.25,
88
+ "max_seq_len": 8192,
89
+ "window_pattern": "L",
90
+ "device_batch_size": 4,
91
+ "total_batch_size": 524288,
92
+ "save_every": 500,
93
+ "eval_every": 250,
94
+ "eval_tokens": 2097152,
95
+ "core_metric_every": -1,
96
+ "sample_every": -1
97
+ },
98
+ "artifacts": {
99
+ "repo": "jbduran/think.nano"
100
+ },
101
+ "wandb": {
102
+ "entity": "jbduran-thinkingmachinesncsu",
103
+ "project": "think.nano",
104
+ "name": "think-d12-r11.25-ctx8192",
105
+ "group": "think-d12",
106
+ "tags": [
107
+ "think-dataset",
108
+ "d12",
109
+ "ratio11.25",
110
+ "ctx8192"
111
+ ]
112
+ },
113
+ "config_fingerprint": "407a5074e0bf3730",
114
+ "artifact_path": "experiments/think-d12-r11.25-ctx8192"
115
+ },
116
+ "stage": "base",
117
+ "base_experiment_id": "think-d12-r11.25-ctx8192",
118
+ "parent_experiment_id": null,
119
+ "parent_checkpoint_step": null,
120
+ "config_fingerprint": "407a5074e0bf3730"
121
+ },
122
+ "device_batch_size": 4,
123
+ "max_seq_len": 8192,
124
+ "total_batch_size": 524288,
125
+ "dataloader_state_dict": {
126
+ "file_idx": 12,
127
+ "pos": 38471586,
128
+ "epoch": 1,
129
+ "pq_idx": 12,
130
+ "rg_idx": 38471586
131
+ },
132
+ "loop_state": {
133
+ "min_val_bpb": 1.0395519592251896,
134
+ "smooth_train_loss": 3.0155535492313272,
135
+ "total_training_time": 9008.03685593605,
136
+ "stage_training_flops": 1939996919061282816,
137
+ "inherited_parent_flops": 0.0,
138
+ "cumulative_pipeline_training_flops": 1939996919061282816
139
+ }
140
+ }
experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_000500.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:51922864ab7ccea4cd40458bb60898167d02696982c1e4c0de684995e5f26290
3
+ size 792761690
experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001000.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8c23459c7bef825a21a0236f2b9e2efba66c8b427c78a7a6cca852df957fd0e
3
+ size 792761690
experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001500.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5dd6cc545ba54787b342e190d75e857872d1964f10a4032ab0d49640012f84b7
3
+ size 792761690
experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002000.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d6e6c50126914bc4d20f9db6222891ee9ee61c634634b023a13e5f1e583e9403
3
+ size 792761690
experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002362.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:43a07973a6d23613f28d1b43623e06bba393623fd88452e1c495bf26aa9ac4d6
3
+ size 792761690
experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_000500_rank0.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f402cf80bf9d4b917ae2a99240b0ea34cd842f025927c148bccf384ce9b744b8
3
+ size 1246165357
experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001000_rank0.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9ef8f97be5f4890871d5586afd42bc946b9058ae996deaa83c14c8f71610deb9
3
+ size 1246165357
experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001500_rank0.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:df9bf4a59d46c126bd65f5cdc6de8fb531c492fabf70f9db22a5ac27bd08dd2f
3
+ size 1246165357
experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002000_rank0.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f95fc4fe7c0a841f3e443102d00495e4f4305ea955ad2b81ffd3ad8865cbd34e
3
+ size 1246165357
experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002362_rank0.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9e1464e581c921562375dd5957b76dad9fe37d04a17e9a021a7876026e624044
3
+ size 1246165357
experiments/think-d12-r11.25-ctx8192/config.json ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": 1,
3
+ "stage": "base",
4
+ "experiment_id": "think-d12-r11.25-ctx8192",
5
+ "dataset": {
6
+ "adapter": "parquet_shards",
7
+ "repo": "jbduran/think-dataset",
8
+ "revision": "main",
9
+ "validation_shard": 472,
10
+ "num_train_shards": 24,
11
+ "download_workers": 4
12
+ },
13
+ "tokenizer": {
14
+ "mode": "train",
15
+ "max_chars": 2000000000,
16
+ "doc_cap": 10000,
17
+ "vocab_size": 32768
18
+ },
19
+ "pretokenize": {
20
+ "enabled": true,
21
+ "slack": 1.03,
22
+ "val_tokens": 20971520,
23
+ "shard_tokens": 100000000,
24
+ "tokenizer_threads": 8
25
+ },
26
+ "training": {
27
+ "depth": 12,
28
+ "scaling_params": 110100912,
29
+ "target_param_data_ratio": 11.25,
30
+ "max_seq_len": 8192,
31
+ "window_pattern": "L",
32
+ "device_batch_size": 4,
33
+ "total_batch_size": 524288,
34
+ "save_every": 500,
35
+ "eval_every": 250,
36
+ "eval_tokens": 2097152,
37
+ "core_metric_every": -1,
38
+ "sample_every": -1
39
+ },
40
+ "artifacts": {
41
+ "repo": "jbduran/think.nano"
42
+ },
43
+ "wandb": {
44
+ "entity": "jbduran-thinkingmachinesncsu",
45
+ "project": "think.nano",
46
+ "name": "think-d12-r11.25-ctx8192",
47
+ "group": "think-d12",
48
+ "tags": [
49
+ "think-dataset",
50
+ "d12",
51
+ "ratio11.25",
52
+ "ctx8192"
53
+ ]
54
+ },
55
+ "config_fingerprint": "407a5074e0bf3730",
56
+ "artifact_path": "experiments/think-d12-r11.25-ctx8192"
57
+ }
experiments/think-d12-r11.25-ctx8192/evals/core.json ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "base_model (step 2362)",
3
+ "step": 2362,
4
+ "bpb": {},
5
+ "core_metric": 0.07251341817663091,
6
+ "core_results": {
7
+ "hellaswag_zeroshot": 0.2757418751716614,
8
+ "jeopardy": 0.0009447330958209932,
9
+ "bigbench_qa_wikidata": 0.0832144096493721,
10
+ "arc_easy": 0.3194444477558136,
11
+ "arc_challenge": 0.21501706540584564,
12
+ "copa": 0.5099999904632568,
13
+ "commonsense_qa": 0.31285831332206726,
14
+ "piqa": 0.5331882238388062,
15
+ "openbook_qa": 0.24800001084804535,
16
+ "lambada_openai": 0.23423248529434204,
17
+ "hellaswag": 0.2802230417728424,
18
+ "winograd": 0.5714285969734192,
19
+ "winogrande": 0.4956590235233307,
20
+ "bigbench_dyck_languages": 0.10200000554323196,
21
+ "agi_eval_lsat_ar": 0.260869562625885,
22
+ "bigbench_cs_algorithms": 0.41969695687294006,
23
+ "bigbench_operators": 0.07619047909975052,
24
+ "bigbench_repeat_copy_logic": 0.0,
25
+ "squad": 0.024030273780226707,
26
+ "coqa": 0.0821746215224266,
27
+ "boolq": 0.5590214133262634,
28
+ "bigbench_language_identification": 0.2524999976158142
29
+ },
30
+ "centered_results": {
31
+ "hellaswag_zeroshot": 0.034322500228881836,
32
+ "jeopardy": 0.0009447330958209932,
33
+ "bigbench_qa_wikidata": 0.0832144096493721,
34
+ "arc_easy": 0.09259259700775146,
35
+ "arc_challenge": -0.04664391279220581,
36
+ "copa": 0.019999980926513672,
37
+ "commonsense_qa": 0.14107289165258405,
38
+ "piqa": 0.0663764476776123,
39
+ "openbook_qa": -0.002666652202606201,
40
+ "lambada_openai": 0.23423248529434204,
41
+ "hellaswag": 0.04029738903045654,
42
+ "winograd": 0.14285719394683838,
43
+ "winogrande": -0.008681952953338623,
44
+ "bigbench_dyck_languages": 0.10200000554323196,
45
+ "agi_eval_lsat_ar": 0.07608695328235625,
46
+ "bigbench_cs_algorithms": 0.41969695687294006,
47
+ "bigbench_operators": 0.07619047909975052,
48
+ "bigbench_repeat_copy_logic": 0.0,
49
+ "squad": 0.024030273780226707,
50
+ "coqa": 0.0821746215224266,
51
+ "boolq": -0.1604699649308857,
52
+ "bigbench_language_identification": 0.177667764153811
53
+ },
54
+ "conditioned_samples": [],
55
+ "unconditioned_samples": []
56
+ }
experiments/think-d12-r11.25-ctx8192/evals/samples.json ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "base_model (step 2362)",
3
+ "step": 2362,
4
+ "bpb": {},
5
+ "core_metric": null,
6
+ "core_results": null,
7
+ "centered_results": null,
8
+ "conditioned_samples": [
9
+ {
10
+ "prompt": "The capital of France is",
11
+ "text": "<|bos|>The capital of France is not yet fully developed. The capital of the United States is not yet fully developed"
12
+ },
13
+ {
14
+ "prompt": "The chemical symbol of gold is",
15
+ "text": "<|bos|>The chemical symbol of gold is the symbol of the gold, and the symbol of the silver. The gold is"
16
+ },
17
+ {
18
+ "prompt": "If yesterday was Friday, then tomorrow will be",
19
+ "text": "<|bos|>If yesterday was Friday, then tomorrow will be the day of the week. \n\nThe day of the week is the same as"
20
+ },
21
+ {
22
+ "prompt": "The opposite of hot is",
23
+ "text": "<|bos|>The opposite of hot is the same as hot. \n\nThe hot is the same as hot. \n\nThe"
24
+ },
25
+ {
26
+ "prompt": "The planets of the solar system are:",
27
+ "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, 2. The moon, 3. The"
28
+ },
29
+ {
30
+ "prompt": "My favorite color is",
31
+ "text": "<|bos|>My favorite color is the color of the skin of the face, and the color of the skin."
32
+ },
33
+ {
34
+ "prompt": "If 5*x + 3 = 13, then x is",
35
+ "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the"
36
+ }
37
+ ],
38
+ "unconditioned_samples": [
39
+ "<|bos|>ALIENS. A concern called industrial, in which all miners of ability for useful labor were engaged in obtaining industrial materials for making snuffers. Although it would be a rude and untenable enterprise to make different classes of miners dispose of goods for profit, different miners differing between the quality of the material misspelled and its quantity, it always staggers the mind with the idea of the matter which it concerns.\u00b9 Four or five shopkeepers are seen at so many tradeshops in town near together in towns and villages. \n\nMoney is better paid to supply the needs of skilled men than it is in overcrowded",
40
+ "<|bos|>37374.31 617.471.11 379.75 \n\nSwinburne, Jes. 10, 335.\n\nStatistical Index. \n\nCoates, erance, 1333. \n\nPurple Debenture, 1460.\n\nGenetic Index. \n\nSwinfenning, 51858.29, 1319. \n\nFree Presses. \n\nRidgens-Pantrepous, 25. \n\nSprayl-Power, regular exercise, 1700. \n\nPreachers and Teachers of the Schools, 401\u20132. \n\n",
41
+ "<|bos|>URE FOOD AND THE BODY' \n\nIn such a commonwealth Siamese readers might find in Kumber's Essays, or Mabon's Vol. of Tobit and St. Jerome's Lives, a sound, a clear and satisfactory explanation of this phrase. May Lady Cassius inform the reverend Society from which this paragraph is borrowed that in this hour of peril men and women may \"paint to him,\" and bewail Rest and Treatment. In both of these circles there are three main meanings attached to the phrase. One is, with the excessive reference in hotel-keepers, donkeys, or hares; another, with alligators",
42
+ "<|bos|>HENRY MART 140. \n\nLeblay, Mr. De Martyn's invention of music, 4. IX.\n\nTRANSLATOR'S NOTE.-Send forth a translation of the Notes which were received by me, translated from the Musical \n\nCommission's Calendar.\n\nHis performance we cannot altogether estimate, but St. Columba gave us confirmation of his inventions, 30. XXVIIii, 18. Eh, What (Georgics, I, 177); 'Slightest book that ever was written' (Sonn., lies 28\u00bd), a work of high merit read with honour,\n\nQu\u00e6rese",
43
+ "<|bos|>Harvard School, IV Department of Education, 1843-1972. \n\n2 Henry State League, LL. concerning Courses in Medicine and the Arts, pp. 22 et seq.\n\nHistory of the Monroe Doctrine, by one who has visited Europe, compiled from European Authorities.\n\nNew York: N. Y. \n\nExaminer, Vol. XXXI, \n\nApril, 1917, p. 88.\n\nPamphlet on Scien tific Methods of Education (\"Outlines of the Maladies and Defects of the Methods of Industrial Society,\" by Dr. Jevons). \n\nNew York, September and October, 19",
44
+ "<|bos|>The Bird reflects upon. his. \n\nThe Clipper. \n\nVapour.\n\nUntil within a few weeks the admission of the truth to our beloved Bird was fatal to that race, little cared for neither in her recorded history nor since they married, and still less as regards her character. Her reign ended; and when she died only after a few months good for nothing the country felt herself well restored to health. She began now to see her way. Our dear bird became as dear to her as the Christian mother; she began to see her way clearer to her senses; and as her thoughts turned, and freedom fell back, she",
45
+ "<|bos|>Army of the Cumberland and Arkansas Army, and an Army of the Potomac under the command of Martin Robertson.\n\nHEADQUARTERS CAMP THIRDQUARTERS, THIRD BRIG 1ST BRIG 1ST BRIG 1ST BRIG 1ST BRIG \n\n6 8 8 \n\nMCLQUERISHER'S STATION, 9 P.M. \n\nMY ARMY, CAL. \n\nEnlarged with orders by the War Department.\n\nHeadquarters Camp War Department, Clope Ridge, Va., September 15, 1864. \n\n6 P.M The Confederates tend S. M. Camp are in the Confederate service hospital at N. C.",
46
+ "<|bos|>the eightieth year of his age.\n\nI had never been conferring, like the palette and the gallens at which I used to sit. never had thought it wrong to give the faintest hint of this prudery, which I trust is always requested of the upholder at his housekeeping, as shall appear by the order and directions accompanying it. So while I was listening to the wise old voice of the tender bride calling alone in her measure that nonsense of patriarchal impiety. It was all the more gratifying when I heard that the Governor of Shetton is now apparently labouring in the same breath, when he speaks"
47
+ ]
48
+ }
experiments/think-d12-r11.25-ctx8192/evals/val_bpb.json ADDED
@@ -0,0 +1,174 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "base_model (step 2362)",
3
+ "step": 2362,
4
+ "bpb": {
5
+ "val_per_position": [
6
+ {
7
+ "start": 0,
8
+ "end": 256,
9
+ "bpb": 1.1345637945630735
10
+ },
11
+ {
12
+ "start": 256,
13
+ "end": 512,
14
+ "bpb": 1.0678589586934193
15
+ },
16
+ {
17
+ "start": 512,
18
+ "end": 768,
19
+ "bpb": 1.0513444442729074
20
+ },
21
+ {
22
+ "start": 768,
23
+ "end": 1024,
24
+ "bpb": 1.0420207610099703
25
+ },
26
+ {
27
+ "start": 1024,
28
+ "end": 1280,
29
+ "bpb": 1.0363256199642377
30
+ },
31
+ {
32
+ "start": 1280,
33
+ "end": 1536,
34
+ "bpb": 1.0293503892901525
35
+ },
36
+ {
37
+ "start": 1536,
38
+ "end": 1792,
39
+ "bpb": 1.0270513457476949
40
+ },
41
+ {
42
+ "start": 1792,
43
+ "end": 2048,
44
+ "bpb": 1.0255301775348173
45
+ },
46
+ {
47
+ "start": 2048,
48
+ "end": 2304,
49
+ "bpb": 1.0171632048394768
50
+ },
51
+ {
52
+ "start": 2304,
53
+ "end": 2560,
54
+ "bpb": 1.0150940036004366
55
+ },
56
+ {
57
+ "start": 2560,
58
+ "end": 2816,
59
+ "bpb": 1.0132597834490384
60
+ },
61
+ {
62
+ "start": 2816,
63
+ "end": 3072,
64
+ "bpb": 1.0154303287433495
65
+ },
66
+ {
67
+ "start": 3072,
68
+ "end": 3328,
69
+ "bpb": 1.0125085420227948
70
+ },
71
+ {
72
+ "start": 3328,
73
+ "end": 3584,
74
+ "bpb": 1.0117220799135607
75
+ },
76
+ {
77
+ "start": 3584,
78
+ "end": 3840,
79
+ "bpb": 1.0093052242865306
80
+ },
81
+ {
82
+ "start": 3840,
83
+ "end": 4096,
84
+ "bpb": 1.0067290549863526
85
+ },
86
+ {
87
+ "start": 4096,
88
+ "end": 4352,
89
+ "bpb": 1.007627760298138
90
+ },
91
+ {
92
+ "start": 4352,
93
+ "end": 4608,
94
+ "bpb": 1.005475773581573
95
+ },
96
+ {
97
+ "start": 4608,
98
+ "end": 4864,
99
+ "bpb": 1.0011348016292028
100
+ },
101
+ {
102
+ "start": 4864,
103
+ "end": 5120,
104
+ "bpb": 1.0025369118565095
105
+ },
106
+ {
107
+ "start": 5120,
108
+ "end": 5376,
109
+ "bpb": 0.9978363303279962
110
+ },
111
+ {
112
+ "start": 5376,
113
+ "end": 5632,
114
+ "bpb": 0.9936016054109623
115
+ },
116
+ {
117
+ "start": 5632,
118
+ "end": 5888,
119
+ "bpb": 0.9930789448128515
120
+ },
121
+ {
122
+ "start": 5888,
123
+ "end": 6144,
124
+ "bpb": 0.988120986726172
125
+ },
126
+ {
127
+ "start": 6144,
128
+ "end": 6400,
129
+ "bpb": 0.9878455863729801
130
+ },
131
+ {
132
+ "start": 6400,
133
+ "end": 6656,
134
+ "bpb": 0.988322568304946
135
+ },
136
+ {
137
+ "start": 6656,
138
+ "end": 6912,
139
+ "bpb": 0.9895607000315525
140
+ },
141
+ {
142
+ "start": 6912,
143
+ "end": 7168,
144
+ "bpb": 0.9924135354945043
145
+ },
146
+ {
147
+ "start": 7168,
148
+ "end": 7424,
149
+ "bpb": 0.9887797597227126
150
+ },
151
+ {
152
+ "start": 7424,
153
+ "end": 7680,
154
+ "bpb": 0.9863644464805957
155
+ },
156
+ {
157
+ "start": 7680,
158
+ "end": 7936,
159
+ "bpb": 0.9852729515784135
160
+ },
161
+ {
162
+ "start": 7936,
163
+ "end": 8192,
164
+ "bpb": 0.9838855089135268
165
+ }
166
+ ],
167
+ "val": 1.0127322889400117
168
+ },
169
+ "core_metric": null,
170
+ "core_results": null,
171
+ "centered_results": null,
172
+ "conditioned_samples": [],
173
+ "unconditioned_samples": []
174
+ }
experiments/think-d12-r11.25-ctx8192/run.json ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "experiment_id": "think-d12-r11.25-ctx8192",
3
+ "stage": "base",
4
+ "base_experiment_id": "think-d12-r11.25-ctx8192",
5
+ "parent_experiment_id": null,
6
+ "parent_checkpoint_step": null,
7
+ "config_fingerprint": "407a5074e0bf3730",
8
+ "wandb_run_id": "e3483a4b",
9
+ "created_at": 1783693257
10
+ }
experiments/think-d12-r11.25-ctx8192/summary.json ADDED
@@ -0,0 +1,69 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "experiment_id": "think-d12-r11.25-ctx8192",
3
+ "stage": "base",
4
+ "base_experiment_id": "think-d12-r11.25-ctx8192",
5
+ "parent_experiment_id": null,
6
+ "parent_checkpoint_step": null,
7
+ "dataset": "jbduran/think-dataset",
8
+ "dataset_revision": "main",
9
+ "step": 2362,
10
+ "depth": 12,
11
+ "target_param_data_ratio": 11.25,
12
+ "training_tokens": 1238368256,
13
+ "final_sampled_val_bpb": 1.0395519592251896,
14
+ "minimum_sampled_val_bpb": 1.0395519592251896,
15
+ "full_val_bpb": 1.0127322889400117,
16
+ "core_metric": null,
17
+ "centered_results": null,
18
+ "conditioned_samples": [
19
+ {
20
+ "prompt": "The capital of France is",
21
+ "text": "<|bos|>The capital of France is not yet fully developed. The capital of the United States is not yet fully developed"
22
+ },
23
+ {
24
+ "prompt": "The chemical symbol of gold is",
25
+ "text": "<|bos|>The chemical symbol of gold is the symbol of the gold, and the symbol of the silver. The gold is"
26
+ },
27
+ {
28
+ "prompt": "If yesterday was Friday, then tomorrow will be",
29
+ "text": "<|bos|>If yesterday was Friday, then tomorrow will be the day of the week. \n\nThe day of the week is the same as"
30
+ },
31
+ {
32
+ "prompt": "The opposite of hot is",
33
+ "text": "<|bos|>The opposite of hot is the same as hot. \n\nThe hot is the same as hot. \n\nThe"
34
+ },
35
+ {
36
+ "prompt": "The planets of the solar system are:",
37
+ "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, 2. The moon, 3. The"
38
+ },
39
+ {
40
+ "prompt": "My favorite color is",
41
+ "text": "<|bos|>My favorite color is the color of the skin of the face, and the color of the skin."
42
+ },
43
+ {
44
+ "prompt": "If 5*x + 3 = 13, then x is",
45
+ "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the"
46
+ }
47
+ ],
48
+ "unconditioned_samples": [
49
+ "<|bos|>ALIENS. A concern called industrial, in which all miners of ability for useful labor were engaged in obtaining industrial materials for making snuffers. Although it would be a rude and untenable enterprise to make different classes of miners dispose of goods for profit, different miners differing between the quality of the material misspelled and its quantity, it always staggers the mind with the idea of the matter which it concerns.\u00b9 Four or five shopkeepers are seen at so many tradeshops in town near together in towns and villages. \n\nMoney is better paid to supply the needs of skilled men than it is in overcrowded",
50
+ "<|bos|>37374.31 617.471.11 379.75 \n\nSwinburne, Jes. 10, 335.\n\nStatistical Index. \n\nCoates, erance, 1333. \n\nPurple Debenture, 1460.\n\nGenetic Index. \n\nSwinfenning, 51858.29, 1319. \n\nFree Presses. \n\nRidgens-Pantrepous, 25. \n\nSprayl-Power, regular exercise, 1700. \n\nPreachers and Teachers of the Schools, 401\u20132. \n\n",
51
+ "<|bos|>URE FOOD AND THE BODY' \n\nIn such a commonwealth Siamese readers might find in Kumber's Essays, or Mabon's Vol. of Tobit and St. Jerome's Lives, a sound, a clear and satisfactory explanation of this phrase. May Lady Cassius inform the reverend Society from which this paragraph is borrowed that in this hour of peril men and women may \"paint to him,\" and bewail Rest and Treatment. In both of these circles there are three main meanings attached to the phrase. One is, with the excessive reference in hotel-keepers, donkeys, or hares; another, with alligators",
52
+ "<|bos|>HENRY MART 140. \n\nLeblay, Mr. De Martyn's invention of music, 4. IX.\n\nTRANSLATOR'S NOTE.-Send forth a translation of the Notes which were received by me, translated from the Musical \n\nCommission's Calendar.\n\nHis performance we cannot altogether estimate, but St. Columba gave us confirmation of his inventions, 30. XXVIIii, 18. Eh, What (Georgics, I, 177); 'Slightest book that ever was written' (Sonn., lies 28\u00bd), a work of high merit read with honour,\n\nQu\u00e6rese",
53
+ "<|bos|>Harvard School, IV Department of Education, 1843-1972. \n\n2 Henry State League, LL. concerning Courses in Medicine and the Arts, pp. 22 et seq.\n\nHistory of the Monroe Doctrine, by one who has visited Europe, compiled from European Authorities.\n\nNew York: N. Y. \n\nExaminer, Vol. XXXI, \n\nApril, 1917, p. 88.\n\nPamphlet on Scien tific Methods of Education (\"Outlines of the Maladies and Defects of the Methods of Industrial Society,\" by Dr. Jevons). \n\nNew York, September and October, 19",
54
+ "<|bos|>The Bird reflects upon. his. \n\nThe Clipper. \n\nVapour.\n\nUntil within a few weeks the admission of the truth to our beloved Bird was fatal to that race, little cared for neither in her recorded history nor since they married, and still less as regards her character. Her reign ended; and when she died only after a few months good for nothing the country felt herself well restored to health. She began now to see her way. Our dear bird became as dear to her as the Christian mother; she began to see her way clearer to her senses; and as her thoughts turned, and freedom fell back, she",
55
+ "<|bos|>Army of the Cumberland and Arkansas Army, and an Army of the Potomac under the command of Martin Robertson.\n\nHEADQUARTERS CAMP THIRDQUARTERS, THIRD BRIG 1ST BRIG 1ST BRIG 1ST BRIG 1ST BRIG \n\n6 8 8 \n\nMCLQUERISHER'S STATION, 9 P.M. \n\nMY ARMY, CAL. \n\nEnlarged with orders by the War Department.\n\nHeadquarters Camp War Department, Clope Ridge, Va., September 15, 1864. \n\n6 P.M The Confederates tend S. M. Camp are in the Confederate service hospital at N. C.",
56
+ "<|bos|>the eightieth year of his age.\n\nI had never been conferring, like the palette and the gallens at which I used to sit. never had thought it wrong to give the faintest hint of this prudery, which I trust is always requested of the upholder at his housekeeping, as shall appear by the order and directions accompanying it. So while I was listening to the wise old voice of the tender bride calling alone in her measure that nonsense of patriarchal impiety. It was all the more gratifying when I heard that the Governor of Shetton is now apparently labouring in the same breath, when he speaks"
57
+ ],
58
+ "training_time_seconds": 9008.03685593605,
59
+ "stage_training_flops": 1.9399969190612828e+18,
60
+ "inherited_parent_flops": 0.0,
61
+ "cumulative_pipeline_training_flops": 1.9399969190612828e+18,
62
+ "config_fingerprint": "407a5074e0bf3730",
63
+ "git_commit_sha": "083cd7f99484b5e894a23e4a093de6d339c412ea",
64
+ "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/e3483a4b",
65
+ "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r11.25-ctx8192",
66
+ "dataset_fingerprint": "63a5e6be81591d82",
67
+ "tokenizer_fingerprint": "03c4f62e7a9d0c3b",
68
+ "unique_train_tokens": 0
69
+ }
experiments/think-d12-r11.25-ctx8192/tokenizer/experiment_tokenizer.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "experiment_id": "think-d12-r11.25-ctx8192",
3
+ "dataset": {
4
+ "adapter": "parquet_shards",
5
+ "repo": "jbduran/think-dataset",
6
+ "revision": "main",
7
+ "validation_shard": 472,
8
+ "num_train_shards": 24,
9
+ "download_workers": 4
10
+ },
11
+ "tokenizer": {
12
+ "mode": "train",
13
+ "max_chars": 2000000000,
14
+ "doc_cap": 10000,
15
+ "vocab_size": 32768
16
+ },
17
+ "created_at": 1783708207
18
+ }
experiments/think-d12-r11.25-ctx8192/tokenizer/token_bytes.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1
3
+ size 132649
experiments/think-d12-r11.25-ctx8192/tokenizer/tokenizer.pkl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1
3
+ size 404071
experiments/think-d12-r11.25/base_checkpoints/meta_000500.json ADDED
@@ -0,0 +1,61 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "step": 500,
3
+ "val_bpb": null,
4
+ "model_config": {
5
+ "sequence_len": 2048,
6
+ "vocab_size": 32768,
7
+ "n_layer": 12,
8
+ "n_head": 6,
9
+ "n_kv_head": 6,
10
+ "n_embd": 768,
11
+ "window_pattern": "L"
12
+ },
13
+ "user_config": {
14
+ "run": "dummy",
15
+ "device_type": "",
16
+ "fp8": false,
17
+ "fp8_recipe": "tensorwise",
18
+ "depth": 12,
19
+ "aspect_ratio": 64,
20
+ "head_dim": 128,
21
+ "max_seq_len": 2048,
22
+ "window_pattern": "L",
23
+ "num_iterations": -1,
24
+ "target_flops": -1.0,
25
+ "target_param_data_ratio": 11.25,
26
+ "device_batch_size": 16,
27
+ "total_batch_size": -1,
28
+ "embedding_lr": 0.3,
29
+ "unembedding_lr": 0.008,
30
+ "weight_decay": 0.28,
31
+ "matrix_lr": 0.02,
32
+ "scalar_lr": 0.5,
33
+ "warmup_steps": 40,
34
+ "warmdown_ratio": 0.65,
35
+ "final_lr_frac": 0.05,
36
+ "resume_from_step": -1,
37
+ "pretokenized": true,
38
+ "eval_every": -1,
39
+ "eval_tokens": 41943040,
40
+ "core_metric_every": -1,
41
+ "core_metric_max_per_task": 500,
42
+ "sample_every": -1,
43
+ "save_every": 500,
44
+ "model_tag": null
45
+ },
46
+ "device_batch_size": 16,
47
+ "max_seq_len": 2048,
48
+ "total_batch_size": 524288,
49
+ "dataloader_state_dict": {
50
+ "file_idx": 2,
51
+ "pos": 62184769,
52
+ "epoch": 1,
53
+ "pq_idx": 2,
54
+ "rg_idx": 62184769
55
+ },
56
+ "loop_state": {
57
+ "min_val_bpb": Infinity,
58
+ "smooth_train_loss": 3.5893779623775406,
59
+ "total_training_time": 1290.4532148838043
60
+ }
61
+ }
experiments/think-d12-r11.25/base_checkpoints/meta_001000.json ADDED
@@ -0,0 +1,61 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "step": 1000,
3
+ "val_bpb": null,
4
+ "model_config": {
5
+ "sequence_len": 2048,
6
+ "vocab_size": 32768,
7
+ "n_layer": 12,
8
+ "n_head": 6,
9
+ "n_kv_head": 6,
10
+ "n_embd": 768,
11
+ "window_pattern": "L"
12
+ },
13
+ "user_config": {
14
+ "run": "dummy",
15
+ "device_type": "",
16
+ "fp8": false,
17
+ "fp8_recipe": "tensorwise",
18
+ "depth": 12,
19
+ "aspect_ratio": 64,
20
+ "head_dim": 128,
21
+ "max_seq_len": 2048,
22
+ "window_pattern": "L",
23
+ "num_iterations": -1,
24
+ "target_flops": -1.0,
25
+ "target_param_data_ratio": 11.25,
26
+ "device_batch_size": 16,
27
+ "total_batch_size": -1,
28
+ "embedding_lr": 0.3,
29
+ "unembedding_lr": 0.008,
30
+ "weight_decay": 0.28,
31
+ "matrix_lr": 0.02,
32
+ "scalar_lr": 0.5,
33
+ "warmup_steps": 40,
34
+ "warmdown_ratio": 0.65,
35
+ "final_lr_frac": 0.05,
36
+ "resume_from_step": -1,
37
+ "pretokenized": true,
38
+ "eval_every": -1,
39
+ "eval_tokens": 41943040,
40
+ "core_metric_every": -1,
41
+ "core_metric_max_per_task": 500,
42
+ "sample_every": -1,
43
+ "save_every": 500,
44
+ "model_tag": null
45
+ },
46
+ "device_batch_size": 16,
47
+ "max_seq_len": 2048,
48
+ "total_batch_size": 524288,
49
+ "dataloader_state_dict": {
50
+ "file_idx": 5,
51
+ "pos": 24336769,
52
+ "epoch": 1,
53
+ "pq_idx": 5,
54
+ "rg_idx": 24336769
55
+ },
56
+ "loop_state": {
57
+ "min_val_bpb": Infinity,
58
+ "smooth_train_loss": 3.4419165825253133,
59
+ "total_training_time": 2609.577807664871
60
+ }
61
+ }
experiments/think-d12-r11.25/base_checkpoints/meta_001500.json ADDED
@@ -0,0 +1,61 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "step": 1500,
3
+ "val_bpb": null,
4
+ "model_config": {
5
+ "sequence_len": 2048,
6
+ "vocab_size": 32768,
7
+ "n_layer": 12,
8
+ "n_head": 6,
9
+ "n_kv_head": 6,
10
+ "n_embd": 768,
11
+ "window_pattern": "L"
12
+ },
13
+ "user_config": {
14
+ "run": "dummy",
15
+ "device_type": "",
16
+ "fp8": false,
17
+ "fp8_recipe": "tensorwise",
18
+ "depth": 12,
19
+ "aspect_ratio": 64,
20
+ "head_dim": 128,
21
+ "max_seq_len": 2048,
22
+ "window_pattern": "L",
23
+ "num_iterations": -1,
24
+ "target_flops": -1.0,
25
+ "target_param_data_ratio": 11.25,
26
+ "device_batch_size": 16,
27
+ "total_batch_size": -1,
28
+ "embedding_lr": 0.3,
29
+ "unembedding_lr": 0.008,
30
+ "weight_decay": 0.28,
31
+ "matrix_lr": 0.02,
32
+ "scalar_lr": 0.5,
33
+ "warmup_steps": 40,
34
+ "warmdown_ratio": 0.65,
35
+ "final_lr_frac": 0.05,
36
+ "resume_from_step": -1,
37
+ "pretokenized": true,
38
+ "eval_every": -1,
39
+ "eval_tokens": 41943040,
40
+ "core_metric_every": -1,
41
+ "core_metric_max_per_task": 500,
42
+ "sample_every": -1,
43
+ "save_every": 500,
44
+ "model_tag": null
45
+ },
46
+ "device_batch_size": 16,
47
+ "max_seq_len": 2048,
48
+ "total_batch_size": 524288,
49
+ "dataloader_state_dict": {
50
+ "file_idx": 7,
51
+ "pos": 86488769,
52
+ "epoch": 1,
53
+ "pq_idx": 7,
54
+ "rg_idx": 86488769
55
+ },
56
+ "loop_state": {
57
+ "min_val_bpb": Infinity,
58
+ "smooth_train_loss": 3.2465426140603015,
59
+ "total_training_time": 3929.598204135895
60
+ }
61
+ }
experiments/think-d12-r11.25/base_checkpoints/meta_002000.json ADDED
@@ -0,0 +1,61 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "step": 2000,
3
+ "val_bpb": null,
4
+ "model_config": {
5
+ "sequence_len": 2048,
6
+ "vocab_size": 32768,
7
+ "n_layer": 12,
8
+ "n_head": 6,
9
+ "n_kv_head": 6,
10
+ "n_embd": 768,
11
+ "window_pattern": "L"
12
+ },
13
+ "user_config": {
14
+ "run": "dummy",
15
+ "device_type": "",
16
+ "fp8": false,
17
+ "fp8_recipe": "tensorwise",
18
+ "depth": 12,
19
+ "aspect_ratio": 64,
20
+ "head_dim": 128,
21
+ "max_seq_len": 2048,
22
+ "window_pattern": "L",
23
+ "num_iterations": -1,
24
+ "target_flops": -1.0,
25
+ "target_param_data_ratio": 11.25,
26
+ "device_batch_size": 16,
27
+ "total_batch_size": -1,
28
+ "embedding_lr": 0.3,
29
+ "unembedding_lr": 0.008,
30
+ "weight_decay": 0.28,
31
+ "matrix_lr": 0.02,
32
+ "scalar_lr": 0.5,
33
+ "warmup_steps": 40,
34
+ "warmdown_ratio": 0.65,
35
+ "final_lr_frac": 0.05,
36
+ "resume_from_step": -1,
37
+ "pretokenized": true,
38
+ "eval_every": -1,
39
+ "eval_tokens": 41943040,
40
+ "core_metric_every": -1,
41
+ "core_metric_max_per_task": 500,
42
+ "sample_every": -1,
43
+ "save_every": 500,
44
+ "model_tag": null
45
+ },
46
+ "device_batch_size": 16,
47
+ "max_seq_len": 2048,
48
+ "total_batch_size": 524288,
49
+ "dataloader_state_dict": {
50
+ "file_idx": 10,
51
+ "pos": 48640769,
52
+ "epoch": 1,
53
+ "pq_idx": 10,
54
+ "rg_idx": 48640769
55
+ },
56
+ "loop_state": {
57
+ "min_val_bpb": Infinity,
58
+ "smooth_train_loss": 3.2442641345309373,
59
+ "total_training_time": 5249.494728565216
60
+ }
61
+ }
experiments/think-d12-r11.25/base_checkpoints/meta_002362.json ADDED
@@ -0,0 +1,61 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "step": 2362,
3
+ "val_bpb": null,
4
+ "model_config": {
5
+ "sequence_len": 2048,
6
+ "vocab_size": 32768,
7
+ "n_layer": 12,
8
+ "n_head": 6,
9
+ "n_kv_head": 6,
10
+ "n_embd": 768,
11
+ "window_pattern": "L"
12
+ },
13
+ "user_config": {
14
+ "run": "dummy",
15
+ "device_type": "",
16
+ "fp8": false,
17
+ "fp8_recipe": "tensorwise",
18
+ "depth": 12,
19
+ "aspect_ratio": 64,
20
+ "head_dim": 128,
21
+ "max_seq_len": 2048,
22
+ "window_pattern": "L",
23
+ "num_iterations": -1,
24
+ "target_flops": -1.0,
25
+ "target_param_data_ratio": 11.25,
26
+ "device_batch_size": 16,
27
+ "total_batch_size": -1,
28
+ "embedding_lr": 0.3,
29
+ "unembedding_lr": 0.008,
30
+ "weight_decay": 0.28,
31
+ "matrix_lr": 0.02,
32
+ "scalar_lr": 0.5,
33
+ "warmup_steps": 40,
34
+ "warmdown_ratio": 0.65,
35
+ "final_lr_frac": 0.05,
36
+ "resume_from_step": -1,
37
+ "pretokenized": true,
38
+ "eval_every": -1,
39
+ "eval_tokens": 41943040,
40
+ "core_metric_every": -1,
41
+ "core_metric_max_per_task": 500,
42
+ "sample_every": -1,
43
+ "save_every": 500,
44
+ "model_tag": null
45
+ },
46
+ "device_batch_size": 16,
47
+ "max_seq_len": 2048,
48
+ "total_batch_size": 524288,
49
+ "dataloader_state_dict": {
50
+ "file_idx": 12,
51
+ "pos": 38438817,
52
+ "epoch": 1,
53
+ "pq_idx": 12,
54
+ "rg_idx": 38438817
55
+ },
56
+ "loop_state": {
57
+ "min_val_bpb": Infinity,
58
+ "smooth_train_loss": 3.074421420856799,
59
+ "total_training_time": 6205.646646976471
60
+ }
61
+ }
experiments/think-d12-r11.25/base_checkpoints/model_000500.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:21fbdff366e5db3fa2f254b4b98c763cc70ba95722242fa032d8d21b956b694f
3
+ size 792761399
experiments/think-d12-r11.25/base_checkpoints/model_001000.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6f0a862044e51b8cac250cf1e4f26af7f8347fd887b8c30af08addb879b6494d
3
+ size 792761399
experiments/think-d12-r11.25/base_checkpoints/model_001500.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:326bddc26c100de33310f9164dc873af6099a9b7760527a8707c256988a8ec7a
3
+ size 792761399
experiments/think-d12-r11.25/base_checkpoints/model_002000.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a5d101975722cffc43eb4f35aa967a5afd397b159da3521e5a1acbd3819d466e
3
+ size 792761399
experiments/think-d12-r11.25/base_checkpoints/model_002362.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:04331c52ed8fa7259f350e4ec72d0dd6602451cfd75a5773a4c17ac5c141ea7e
3
+ size 792761399
experiments/think-d12-r11.25/base_checkpoints/optim_000500_rank0.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d39a131089a476a202ae932f21d9398b225d0b8e4147a58fbdff797914d34976
3
+ size 1246165237
experiments/think-d12-r11.25/base_checkpoints/optim_001000_rank0.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e4971566eb3d4aa5d5fe29b3a3b77f56581b61713e5c1d162debb8a409c02118
3
+ size 1246165237
experiments/think-d12-r11.25/base_checkpoints/optim_001500_rank0.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:05ce87d44bfd6e9b4d4c1e643a2a5aa4b169119913ccbdf3625d7cfa7313b342
3
+ size 1246165237
experiments/think-d12-r11.25/base_checkpoints/optim_002000_rank0.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1b58f68fd887360ce99e8456aecf706b1bcf5c8210194721389698ec658237b1
3
+ size 1246165237
experiments/think-d12-r11.25/base_checkpoints/optim_002362_rank0.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:09aa5632a4b33981a1b2fbd97254d0050ecb7916d0c242f1d195c0726be314d7
3
+ size 1246165237
experiments/think-d12-r11.25/config.json ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": 1,
3
+ "stage": "base",
4
+ "experiment_id": "think-d12-r11.25",
5
+ "dataset": {
6
+ "adapter": "parquet_shards",
7
+ "repo": "jbduran/think-dataset",
8
+ "revision": "main",
9
+ "validation_shard": 472,
10
+ "num_train_shards": 24,
11
+ "download_workers": 4
12
+ },
13
+ "tokenizer": {
14
+ "mode": "train",
15
+ "max_chars": 2000000000,
16
+ "doc_cap": 10000,
17
+ "vocab_size": 32768
18
+ },
19
+ "pretokenize": {
20
+ "enabled": true,
21
+ "slack": 1.03,
22
+ "val_tokens": 20971520,
23
+ "shard_tokens": 100000000,
24
+ "tokenizer_threads": 8
25
+ },
26
+ "training": {
27
+ "depth": 12,
28
+ "scaling_params": 110100912,
29
+ "target_param_data_ratio": 11.25,
30
+ "window_pattern": "L",
31
+ "device_batch_size": 16,
32
+ "total_batch_size": 524288,
33
+ "save_every": 500,
34
+ "eval_every": 250,
35
+ "eval_tokens": 2097152,
36
+ "core_metric_every": -1,
37
+ "sample_every": -1
38
+ },
39
+ "artifacts": {
40
+ "repo": "jbduran/think.nano"
41
+ },
42
+ "wandb": {
43
+ "entity": "jbduran-thinkingmachinesncsu",
44
+ "project": "think.nano",
45
+ "name": "think-d12-r11.25",
46
+ "group": "think-d12",
47
+ "tags": [
48
+ "think-dataset",
49
+ "d12",
50
+ "ratio11.25"
51
+ ]
52
+ },
53
+ "config_fingerprint": "3b5a68714770b6af",
54
+ "artifact_path": "experiments/think-d12-r11.25"
55
+ }
experiments/think-d12-r11.25/evals/samples.json ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "base_model (step 2362)",
3
+ "step": 2362,
4
+ "bpb": {},
5
+ "core_metric": null,
6
+ "core_results": null,
7
+ "centered_results": null,
8
+ "conditioned_samples": [
9
+ {
10
+ "prompt": "The capital of France is",
11
+ "text": "<|bos|>The capital of France is 10,000,000 francs, and the capital of the United"
12
+ },
13
+ {
14
+ "prompt": "The chemical symbol of gold is",
15
+ "text": "<|bos|>The chemical symbol of gold is the gold of the \n\nUnited States. It is the gold of the United States"
16
+ },
17
+ {
18
+ "prompt": "If yesterday was Friday, then tomorrow will be",
19
+ "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nThe day is Sunday, and the day is Sunday. \n\nThe"
20
+ },
21
+ {
22
+ "prompt": "The opposite of hot is",
23
+ "text": "<|bos|>The opposite of hot is the opposite of cold. \n\nThe opposite of cold is the opposite of cold."
24
+ },
25
+ {
26
+ "prompt": "The planets of the solar system are:",
27
+ "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, and the stars. \n\n2."
28
+ },
29
+ {
30
+ "prompt": "My favorite color is",
31
+ "text": "<|bos|>My favorite color is the color of the sky. \n\nThe color of the sky is a color of"
32
+ },
33
+ {
34
+ "prompt": "If 5*x + 3 = 13, then x is",
35
+ "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times the number of times the number of times the number of times"
36
+ }
37
+ ],
38
+ "unconditioned_samples": [
39
+ "<|bos|>Monthillahivan-this is worthy, they say, of the glory of God's presence, our Father being now to come again! \n\nDead. The priest who said Lord! so between thy guilty hands exulting and curse shoot; do thy honors sweetly, O Father! touch me not with panic, nor miss me with holy surprise, nor sigh (say) tost away my days, nor desolate them; thou wert sent by heaven, and seen with so much care, that Thou near thy sins didst repair them; how art Thou now to come thus early, and bear unto me so divinely? not",
40
+ "<|bos|>370 \n\nPaxton's Introduction to Education, or the true philosophy of the schools criticised, is differentiated. In Rossetti's plan, as already noted, it formulated in these brief articles, there should be but two different versions of the 73\n\n-flat elements; in Rossetti's, we commonly use the broad current of their general aim. That which has been called a positive and material-is aptly called a negative-was given with corresponding emphasis at three different times, at different epochs, but the THREE are frequently gods regular in origin, and preserved from the violence of adjustment and shift, rather than dormant and deliberate meaning of",
41
+ "<|bos|>ROrepoys and Willieot's book on the Siamese Law. \n\nA Letter from Frank Perrivsky to a Prince of Wales. By Kinsman Keith. \n\nTHE \n\nMAY, 1923. \n\nIn Paper \n\nWith 25 Illustrations. Parts. $2 $9 $14 6 $2.50 \n\nIN POLITICAL IDOLSES. \n\nAmerican Law. By George W. Resting, LL.D., Professor of Political Economy in Princeton University. \n\n$18 $2.50 \n\nSamson Minot, hotel-keeper. $3 $7\n\nTHE SCALE OF SIZE.",
42
+ "<|bos|>HENRY MARTYN SAXON, THE HINDUS CHRISTURIENT. \n\nEDWARD IRVING, of the College of the Anatomy School of \n\nDurham, Surrey.\n\nEDWARD PERCY BAKER, OF ALICE COLLEGE, whose personal appearance is now in print, was born at Stanneley, Surrey, July 18, 1794. He has been student in the Company's Military College at Woolwich for more than one year, for his learning and industry in his profession and studies. He has written a book entitled History, Economics and Political Science, which lies nearly at our very door, entitled History, Political Science and Political \n\nScience. It is",
43
+ "<|bos|> HOUSE OF THE ANGELS. \n\nFrom the City of St. Ann. \n\n2 vols. 3s. My Last in a Garden. I reserve for the fifth edition, in manuscript, a full account of ancient His tory. In \n\n1 vol. 5, a short history of Nero and Herod. from 6 to \n\n10 Years, both kept in this Library.\n\n4 \n\nLibrary of the Inducci EN QUANDRON. \n\nFRONTIER, SAMUEL, Dean of Carlisle, President of the\n\nAmerican Board of Works, 3 vols. 2 vols. 3s. 6d. \n\nLibrary",
44
+ "<|bos|>The Bird reflects the world, and swims the eagle's web.-Van Isle.\n\nSHOP'S \n\nREACH \n\nNEGARYELY GENIAL SHOP FINANCIERS RAIDER \n\nREAR HEADS,.} \n\nREPUBLICS, WHILE SHE UNDER FULLER'S PORT, \n\nREP Philosophers, that they be not \n\nPharaoh's patterns good for nothing, and valiant men for that which is nothing; \n\nCLY VAUS\u00d2 EXOVENT \u03b4\u1f72 \u03bf\u03cd\u03c3\u03b1\u03b9 \u0391\u03b4\u03af\u03b4\u03b5\u03b9ANTA\u03c1, \u1f15\u03b4\u03b1\u03c1\u03c9\u03bd \u03c3\u03bf\u03c5\u03b4\u03bf\u1f7a\u03c2 The Queen eschews",
45
+ "<|bos|>'Eau du Monarch beau ou H\u00f4tel de Voodjiches.\" (The same French Inspecteur :) \"Son jours\n\nA \u00e9t\u00e9 autant \u00e0 tomboi les m\u00eames Premi\u00e8res avec une princesse qui sont d'ombres le miraculeur bien de France le souffrir. Les c\u00f4tes de ceux qui le sont j\u00e9sibres.\" (The French Directors.)\n\neffected these river improvements in effeminate cases, in the The Executive has eschewed the corrupdemell\u00e8 hrs. Nieuw Ga",
46
+ "<|bos|>the eight (250) series.\n\nThe pressure is already great.\n\nashions like the palette de la \n\nLast chapter, however, it may be noted, and it is a most common practice in such a period for small and countless series to transform the palette de la up half at the sauce, and it is a matter of considerable importance to seeing the clothes-holders uniformly the marks that indicate the supply of the palette de la into that respectable width whence their sulky tints are generally returned. \n\nThe colour represents the price of the garment at the time apparently labouring in the work. Delivery is possible"
47
+ ]
48
+ }
experiments/think-d12-r11.25/evals/val_bpb.json ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "base_model (step 2362)",
3
+ "step": 2362,
4
+ "bpb": {
5
+ "val_per_position": [
6
+ {
7
+ "start": 0,
8
+ "end": 256,
9
+ "bpb": 1.1255410237731462
10
+ },
11
+ {
12
+ "start": 256,
13
+ "end": 512,
14
+ "bpb": 1.0636677720059344
15
+ },
16
+ {
17
+ "start": 512,
18
+ "end": 768,
19
+ "bpb": 1.0506832286096848
20
+ },
21
+ {
22
+ "start": 768,
23
+ "end": 1024,
24
+ "bpb": 1.0456638664028364
25
+ },
26
+ {
27
+ "start": 1024,
28
+ "end": 1280,
29
+ "bpb": 1.038324560694976
30
+ },
31
+ {
32
+ "start": 1280,
33
+ "end": 1536,
34
+ "bpb": 1.0330914959872517
35
+ },
36
+ {
37
+ "start": 1536,
38
+ "end": 1792,
39
+ "bpb": 1.0308104068466784
40
+ },
41
+ {
42
+ "start": 1792,
43
+ "end": 2048,
44
+ "bpb": 1.027579795687179
45
+ }
46
+ ],
47
+ "val": 1.0519195678472355
48
+ },
49
+ "core_metric": null,
50
+ "core_results": null,
51
+ "centered_results": null,
52
+ "conditioned_samples": [],
53
+ "unconditioned_samples": []
54
+ }
experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean-1930s.json ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "base_model (step 2362)",
3
+ "step": 2362,
4
+ "bpb": {
5
+ "val": 1.0781689302026238
6
+ },
7
+ "core_metric": null,
8
+ "core_results": null,
9
+ "centered_results": null,
10
+ "conditioned_samples": [],
11
+ "unconditioned_samples": []
12
+ }
experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "base_model (step 2362)",
3
+ "step": 2362,
4
+ "bpb": {
5
+ "val": 1.081385374786877
6
+ },
7
+ "core_metric": null,
8
+ "core_results": null,
9
+ "centered_results": null,
10
+ "conditioned_samples": [],
11
+ "unconditioned_samples": []
12
+ }
experiments/think-d12-r11.25/run.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "experiment_id": "think-d12-r11.25",
3
+ "stage": "base",
4
+ "wandb_run_id": null,
5
+ "migration_note": "Migrated from the pre-lineage repository layout."
6
+ }
experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/meta_001065.json ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "step": 1065,
3
+ "val_bpb": 0.39271438585109175,
4
+ "model_config": {
5
+ "sequence_len": 2048,
6
+ "vocab_size": 32768,
7
+ "n_layer": 12,
8
+ "n_head": 6,
9
+ "n_kv_head": 6,
10
+ "n_embd": 768,
11
+ "window_pattern": "L"
12
+ },
13
+ "user_config": {
14
+ "run": "dummy",
15
+ "device_type": "",
16
+ "model_tag": "d12",
17
+ "model_step": null,
18
+ "load_optimizer": 1,
19
+ "num_iterations": -1,
20
+ "max_seq_len": null,
21
+ "device_batch_size": 8,
22
+ "total_batch_size": null,
23
+ "embedding_lr": null,
24
+ "unembedding_lr": null,
25
+ "matrix_lr": null,
26
+ "init_lr_frac": 0.8,
27
+ "warmup_ratio": 0.0,
28
+ "warmdown_ratio": 0.5,
29
+ "final_lr_frac": 0.0,
30
+ "eval_every": -1,
31
+ "eval_tokens": 20971520,
32
+ "chatcore_every": -1,
33
+ "chatcore_max_cat": -1,
34
+ "chatcore_max_sample": 24,
35
+ "mmlu_epochs": 3,
36
+ "gsm8k_epochs": 4
37
+ }
38
+ }
experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/model_001065.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f1ef8886ea2cbd820baa9bc98368f99673efb1f5fdbd082196d573b291111bda
3
+ size 792761399
experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/optim_001065_rank0.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:db8c9df6e288618ed6362102e29edc3731267b4ff382154a55709b4d5a14f1d4
3
+ size 1246165237