diff --git a/experiments/think-d12-r11.25-ctx4096/tokenizer/token_bytes.pt b/experiments/think-d12-r11.25-ctx4096/tokenizer/token_bytes.pt new file mode 100644 index 0000000000000000000000000000000000000000..01d1ec4aab9e8a7d205c3b3ffbeb8da0e9a62db1 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx4096/tokenizer/token_bytes.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1 +size 132649 diff --git a/experiments/think-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl b/experiments/think-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl new file mode 100644 index 0000000000000000000000000000000000000000..a17bd392980021628053b95d6425fc556aad527a --- /dev/null +++ b/experiments/think-d12-r11.25-ctx4096/tokenizer/tokenizer.pkl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1 +size 404071 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_000500.json b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_000500.json new file mode 100644 index 0000000000000000000000000000000000000000..f90f8873ef7208da04efc8c72725e8cc3a645313 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_000500.json @@ -0,0 +1,140 @@ +{ + "step": 500, + "training_complete": false, + "experiment_id": "think-d12-r11.25-ctx8192", + "val_bpb": 1.2806006574422366, + "model_config": { + "sequence_len": 8192, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r11.25-ctx8192", + "wandb_run_id": "e3483a4b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 8192, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 4, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints", + "experiment_id": "think-d12-r11.25-ctx8192", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json", + "tokenizer_fingerprint": "03c4f62e7a9d0c3b", + "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r11.25-ctx8192", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r11.25-ctx8192", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 24, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "max_seq_len": 8192, + "window_pattern": "L", + "device_batch_size": 4, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r11.25-ctx8192", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11.25", + "ctx8192" + ] + }, + "config_fingerprint": "407a5074e0bf3730", + "artifact_path": "experiments/think-d12-r11.25-ctx8192" + }, + "stage": "base", + "base_experiment_id": "think-d12-r11.25-ctx8192", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "407a5074e0bf3730" + }, + "device_batch_size": 4, + "max_seq_len": 8192, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 2, + "pos": 62184769, + "epoch": 1, + "pq_idx": 2, + "rg_idx": 62184769 + }, + "loop_state": { + "min_val_bpb": 1.2806006574422366, + "smooth_train_loss": 3.5198863114259193, + "total_training_time": 1859.6971344947815, + "stage_training_flops": 410668272451584000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 410668272451584000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001000.json b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001000.json new file mode 100644 index 0000000000000000000000000000000000000000..ea0348fdef439360f2a2aaaac11e65d388f12e6a --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001000.json @@ -0,0 +1,140 @@ +{ + "step": 1000, + "training_complete": false, + "experiment_id": "think-d12-r11.25-ctx8192", + "val_bpb": 1.1673074973592756, + "model_config": { + "sequence_len": 8192, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r11.25-ctx8192", + "wandb_run_id": "e3483a4b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 8192, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 4, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints", + "experiment_id": "think-d12-r11.25-ctx8192", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json", + "tokenizer_fingerprint": "03c4f62e7a9d0c3b", + "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r11.25-ctx8192", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r11.25-ctx8192", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 24, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "max_seq_len": 8192, + "window_pattern": "L", + "device_batch_size": 4, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r11.25-ctx8192", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11.25", + "ctx8192" + ] + }, + "config_fingerprint": "407a5074e0bf3730", + "artifact_path": "experiments/think-d12-r11.25-ctx8192" + }, + "stage": "base", + "base_experiment_id": "think-d12-r11.25-ctx8192", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "407a5074e0bf3730" + }, + "device_batch_size": 4, + "max_seq_len": 8192, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 5, + "pos": 24336769, + "epoch": 1, + "pq_idx": 5, + "rg_idx": 24336769 + }, + "loop_state": { + "min_val_bpb": 1.1673074973592756, + "smooth_train_loss": 3.291130702382907, + "total_training_time": 3762.280524253845, + "stage_training_flops": 821336544903168000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 821336544903168000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001500.json b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001500.json new file mode 100644 index 0000000000000000000000000000000000000000..17245f9d0ddd15b718a855106a0a3cee19388345 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_001500.json @@ -0,0 +1,140 @@ +{ + "step": 1500, + "training_complete": false, + "experiment_id": "think-d12-r11.25-ctx8192", + "val_bpb": 1.108596107059193, + "model_config": { + "sequence_len": 8192, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r11.25-ctx8192", + "wandb_run_id": "e3483a4b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 8192, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 4, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints", + "experiment_id": "think-d12-r11.25-ctx8192", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json", + "tokenizer_fingerprint": "03c4f62e7a9d0c3b", + "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r11.25-ctx8192", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r11.25-ctx8192", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 24, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "max_seq_len": 8192, + "window_pattern": "L", + "device_batch_size": 4, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r11.25-ctx8192", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11.25", + "ctx8192" + ] + }, + "config_fingerprint": "407a5074e0bf3730", + "artifact_path": "experiments/think-d12-r11.25-ctx8192" + }, + "stage": "base", + "base_experiment_id": "think-d12-r11.25-ctx8192", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "407a5074e0bf3730" + }, + "device_batch_size": 4, + "max_seq_len": 8192, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 7, + "pos": 86488769, + "epoch": 1, + "pq_idx": 7, + "rg_idx": 86488769 + }, + "loop_state": { + "min_val_bpb": 1.108596107059193, + "smooth_train_loss": 3.1279163553930784, + "total_training_time": 5663.0339615345, + "stage_training_flops": 1232004817354752000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1232004817354752000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002000.json b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002000.json new file mode 100644 index 0000000000000000000000000000000000000000..dc0ff369001a8ac804a19eebde85f9077dc18a0d --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002000.json @@ -0,0 +1,140 @@ +{ + "step": 2000, + "training_complete": false, + "experiment_id": "think-d12-r11.25-ctx8192", + "val_bpb": 1.0589838913914653, + "model_config": { + "sequence_len": 8192, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r11.25-ctx8192", + "wandb_run_id": "e3483a4b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 8192, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 4, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints", + "experiment_id": "think-d12-r11.25-ctx8192", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json", + "tokenizer_fingerprint": "03c4f62e7a9d0c3b", + "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r11.25-ctx8192", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r11.25-ctx8192", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 24, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "max_seq_len": 8192, + "window_pattern": "L", + "device_batch_size": 4, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r11.25-ctx8192", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11.25", + "ctx8192" + ] + }, + "config_fingerprint": "407a5074e0bf3730", + "artifact_path": "experiments/think-d12-r11.25-ctx8192" + }, + "stage": "base", + "base_experiment_id": "think-d12-r11.25-ctx8192", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "407a5074e0bf3730" + }, + "device_batch_size": 4, + "max_seq_len": 8192, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 10, + "pos": 48640769, + "epoch": 1, + "pq_idx": 10, + "rg_idx": 48640769 + }, + "loop_state": { + "min_val_bpb": 1.0589838913914653, + "smooth_train_loss": 3.116437961262795, + "total_training_time": 7566.1544008255005, + "stage_training_flops": 1642673089806336000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1642673089806336000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002362.json b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002362.json new file mode 100644 index 0000000000000000000000000000000000000000..b9c4c76cb8a461607a426151d0807d6328dcef59 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/meta_002362.json @@ -0,0 +1,140 @@ +{ + "step": 2362, + "training_complete": true, + "experiment_id": "think-d12-r11.25-ctx8192", + "val_bpb": 1.0395519592251896, + "model_config": { + "sequence_len": 8192, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r11.25-ctx8192", + "wandb_run_id": "e3483a4b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio11.25,ctx8192", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 8192, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 4, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": 2000, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/base_checkpoints", + "experiment_id": "think-d12-r11.25-ctx8192", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r11.25-ctx8192/config.json", + "tokenizer_fingerprint": "03c4f62e7a9d0c3b", + "git_commit_sha": "9db3f1c2f998309198af5b020c869fd817d51847", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r11.25-ctx8192", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r11.25-ctx8192", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 24, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "max_seq_len": 8192, + "window_pattern": "L", + "device_batch_size": 4, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r11.25-ctx8192", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11.25", + "ctx8192" + ] + }, + "config_fingerprint": "407a5074e0bf3730", + "artifact_path": "experiments/think-d12-r11.25-ctx8192" + }, + "stage": "base", + "base_experiment_id": "think-d12-r11.25-ctx8192", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "407a5074e0bf3730" + }, + "device_batch_size": 4, + "max_seq_len": 8192, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 12, + "pos": 38471586, + "epoch": 1, + "pq_idx": 12, + "rg_idx": 38471586 + }, + "loop_state": { + "min_val_bpb": 1.0395519592251896, + "smooth_train_loss": 3.0155535492313272, + "total_training_time": 9008.03685593605, + "stage_training_flops": 1939996919061282816, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1939996919061282816 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_000500.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_000500.pt new file mode 100644 index 0000000000000000000000000000000000000000..69f1884a5327e9a0ed0e7a8f939c9443eb75b9e7 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_000500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:51922864ab7ccea4cd40458bb60898167d02696982c1e4c0de684995e5f26290 +size 792761690 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001000.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001000.pt new file mode 100644 index 0000000000000000000000000000000000000000..cba8efe605c7d58052dff025a32ac566cc343a2b --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c8c23459c7bef825a21a0236f2b9e2efba66c8b427c78a7a6cca852df957fd0e +size 792761690 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001500.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001500.pt new file mode 100644 index 0000000000000000000000000000000000000000..3d9fee17f3c145eb8c8c6fd0413a26b73a45da30 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_001500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5dd6cc545ba54787b342e190d75e857872d1964f10a4032ab0d49640012f84b7 +size 792761690 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002000.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002000.pt new file mode 100644 index 0000000000000000000000000000000000000000..3be0f557534f7754337973fd2bdb3d2da5e582db --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d6e6c50126914bc4d20f9db6222891ee9ee61c634634b023a13e5f1e583e9403 +size 792761690 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002362.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002362.pt new file mode 100644 index 0000000000000000000000000000000000000000..0aa26c24cc471452dc63a38e2a0a94011dd47672 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/model_002362.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:43a07973a6d23613f28d1b43623e06bba393623fd88452e1c495bf26aa9ac4d6 +size 792761690 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_000500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..9f7ff257118d85ec5b935356ba451f8158fbdf2f --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_000500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f402cf80bf9d4b917ae2a99240b0ea34cd842f025927c148bccf384ce9b744b8 +size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..477768bd2b6531e7859664b868c3772d4c4fb848 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9ef8f97be5f4890871d5586afd42bc946b9058ae996deaa83c14c8f71610deb9 +size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..7d40d701879db1b27107cbb56c30a19f220e747d --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_001500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:df9bf4a59d46c126bd65f5cdc6de8fb531c492fabf70f9db22a5ac27bd08dd2f +size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..bce34810607710181c5acee791e2beed95a1a712 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f95fc4fe7c0a841f3e443102d00495e4f4305ea955ad2b81ffd3ad8865cbd34e +size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002362_rank0.pt b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002362_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..6e07dc627985d4520e9098d9170ff536446a3daa --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/base_checkpoints/optim_002362_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9e1464e581c921562375dd5957b76dad9fe37d04a17e9a021a7876026e624044 +size 1246165357 diff --git a/experiments/think-d12-r11.25-ctx8192/config.json b/experiments/think-d12-r11.25-ctx8192/config.json new file mode 100644 index 0000000000000000000000000000000000000000..71f8e9d529e0058c9b6ad1539cd12c5d9edb1780 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/config.json @@ -0,0 +1,57 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r11.25-ctx8192", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 24, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "max_seq_len": 8192, + "window_pattern": "L", + "device_batch_size": 4, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r11.25-ctx8192", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11.25", + "ctx8192" + ] + }, + "config_fingerprint": "407a5074e0bf3730", + "artifact_path": "experiments/think-d12-r11.25-ctx8192" +} diff --git a/experiments/think-d12-r11.25-ctx8192/evals/core.json b/experiments/think-d12-r11.25-ctx8192/evals/core.json new file mode 100644 index 0000000000000000000000000000000000000000..feacd2f1e36f41ef00be8749074b5ac013f2579c --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/evals/core.json @@ -0,0 +1,56 @@ +{ + "model": "base_model (step 2362)", + "step": 2362, + "bpb": {}, + "core_metric": 0.07251341817663091, + "core_results": { + "hellaswag_zeroshot": 0.2757418751716614, + "jeopardy": 0.0009447330958209932, + "bigbench_qa_wikidata": 0.0832144096493721, + "arc_easy": 0.3194444477558136, + "arc_challenge": 0.21501706540584564, + "copa": 0.5099999904632568, + "commonsense_qa": 0.31285831332206726, + "piqa": 0.5331882238388062, + "openbook_qa": 0.24800001084804535, + "lambada_openai": 0.23423248529434204, + "hellaswag": 0.2802230417728424, + "winograd": 0.5714285969734192, + "winogrande": 0.4956590235233307, + "bigbench_dyck_languages": 0.10200000554323196, + "agi_eval_lsat_ar": 0.260869562625885, + "bigbench_cs_algorithms": 0.41969695687294006, + "bigbench_operators": 0.07619047909975052, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.024030273780226707, + "coqa": 0.0821746215224266, + "boolq": 0.5590214133262634, + "bigbench_language_identification": 0.2524999976158142 + }, + "centered_results": { + "hellaswag_zeroshot": 0.034322500228881836, + "jeopardy": 0.0009447330958209932, + "bigbench_qa_wikidata": 0.0832144096493721, + "arc_easy": 0.09259259700775146, + "arc_challenge": -0.04664391279220581, + "copa": 0.019999980926513672, + "commonsense_qa": 0.14107289165258405, + "piqa": 0.0663764476776123, + "openbook_qa": -0.002666652202606201, + "lambada_openai": 0.23423248529434204, + "hellaswag": 0.04029738903045654, + "winograd": 0.14285719394683838, + "winogrande": -0.008681952953338623, + "bigbench_dyck_languages": 0.10200000554323196, + "agi_eval_lsat_ar": 0.07608695328235625, + "bigbench_cs_algorithms": 0.41969695687294006, + "bigbench_operators": 0.07619047909975052, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.024030273780226707, + "coqa": 0.0821746215224266, + "boolq": -0.1604699649308857, + "bigbench_language_identification": 0.177667764153811 + }, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/evals/samples.json b/experiments/think-d12-r11.25-ctx8192/evals/samples.json new file mode 100644 index 0000000000000000000000000000000000000000..f01d40c79c96e236727f04968d1a994c68c93574 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/evals/samples.json @@ -0,0 +1,48 @@ +{ + "model": "base_model (step 2362)", + "step": 2362, + "bpb": {}, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is not yet fully developed. The capital of the United States is not yet fully developed" + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is the symbol of the gold, and the symbol of the silver. The gold is" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be the day of the week. \n\nThe day of the week is the same as" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is the same as hot. \n\nThe hot is the same as hot. \n\nThe" + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, 2. The moon, 3. The" + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is the color of the skin of the face, and the color of the skin." + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the" + } + ], + "unconditioned_samples": [ + "<|bos|>ALIENS. A concern called industrial, in which all miners of ability for useful labor were engaged in obtaining industrial materials for making snuffers. Although it would be a rude and untenable enterprise to make different classes of miners dispose of goods for profit, different miners differing between the quality of the material misspelled and its quantity, it always staggers the mind with the idea of the matter which it concerns.\u00b9 Four or five shopkeepers are seen at so many tradeshops in town near together in towns and villages. \n\nMoney is better paid to supply the needs of skilled men than it is in overcrowded", + "<|bos|>37374.31 617.471.11 379.75 \n\nSwinburne, Jes. 10, 335.\n\nStatistical Index. \n\nCoates, erance, 1333. \n\nPurple Debenture, 1460.\n\nGenetic Index. \n\nSwinfenning, 51858.29, 1319. \n\nFree Presses. \n\nRidgens-Pantrepous, 25. \n\nSprayl-Power, regular exercise, 1700. \n\nPreachers and Teachers of the Schools, 401\u20132. \n\n", + "<|bos|>URE FOOD AND THE BODY' \n\nIn such a commonwealth Siamese readers might find in Kumber's Essays, or Mabon's Vol. of Tobit and St. Jerome's Lives, a sound, a clear and satisfactory explanation of this phrase. May Lady Cassius inform the reverend Society from which this paragraph is borrowed that in this hour of peril men and women may \"paint to him,\" and bewail Rest and Treatment. In both of these circles there are three main meanings attached to the phrase. One is, with the excessive reference in hotel-keepers, donkeys, or hares; another, with alligators", + "<|bos|>HENRY MART 140. \n\nLeblay, Mr. De Martyn's invention of music, 4. IX.\n\nTRANSLATOR'S NOTE.-Send forth a translation of the Notes which were received by me, translated from the Musical \n\nCommission's Calendar.\n\nHis performance we cannot altogether estimate, but St. Columba gave us confirmation of his inventions, 30. XXVIIii, 18. Eh, What (Georgics, I, 177); 'Slightest book that ever was written' (Sonn., lies 28\u00bd), a work of high merit read with honour,\n\nQu\u00e6rese", + "<|bos|>Harvard School, IV Department of Education, 1843-1972. \n\n2 Henry State League, LL. concerning Courses in Medicine and the Arts, pp. 22 et seq.\n\nHistory of the Monroe Doctrine, by one who has visited Europe, compiled from European Authorities.\n\nNew York: N. Y. \n\nExaminer, Vol. XXXI, \n\nApril, 1917, p. 88.\n\nPamphlet on Scien tific Methods of Education (\"Outlines of the Maladies and Defects of the Methods of Industrial Society,\" by Dr. Jevons). \n\nNew York, September and October, 19", + "<|bos|>The Bird reflects upon. his. \n\nThe Clipper. \n\nVapour.\n\nUntil within a few weeks the admission of the truth to our beloved Bird was fatal to that race, little cared for neither in her recorded history nor since they married, and still less as regards her character. Her reign ended; and when she died only after a few months good for nothing the country felt herself well restored to health. She began now to see her way. Our dear bird became as dear to her as the Christian mother; she began to see her way clearer to her senses; and as her thoughts turned, and freedom fell back, she", + "<|bos|>Army of the Cumberland and Arkansas Army, and an Army of the Potomac under the command of Martin Robertson.\n\nHEADQUARTERS CAMP THIRDQUARTERS, THIRD BRIG 1ST BRIG 1ST BRIG 1ST BRIG 1ST BRIG \n\n6 8 8 \n\nMCLQUERISHER'S STATION, 9 P.M. \n\nMY ARMY, CAL. \n\nEnlarged with orders by the War Department.\n\nHeadquarters Camp War Department, Clope Ridge, Va., September 15, 1864. \n\n6 P.M The Confederates tend S. M. Camp are in the Confederate service hospital at N. C.", + "<|bos|>the eightieth year of his age.\n\nI had never been conferring, like the palette and the gallens at which I used to sit. never had thought it wrong to give the faintest hint of this prudery, which I trust is always requested of the upholder at his housekeeping, as shall appear by the order and directions accompanying it. So while I was listening to the wise old voice of the tender bride calling alone in her measure that nonsense of patriarchal impiety. It was all the more gratifying when I heard that the Governor of Shetton is now apparently labouring in the same breath, when he speaks" + ] +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/evals/val_bpb.json b/experiments/think-d12-r11.25-ctx8192/evals/val_bpb.json new file mode 100644 index 0000000000000000000000000000000000000000..94e46b7e343d01e00db08cf58ca5b26a3c9e7b5d --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/evals/val_bpb.json @@ -0,0 +1,174 @@ +{ + "model": "base_model (step 2362)", + "step": 2362, + "bpb": { + "val_per_position": [ + { + "start": 0, + "end": 256, + "bpb": 1.1345637945630735 + }, + { + "start": 256, + "end": 512, + "bpb": 1.0678589586934193 + }, + { + "start": 512, + "end": 768, + "bpb": 1.0513444442729074 + }, + { + "start": 768, + "end": 1024, + "bpb": 1.0420207610099703 + }, + { + "start": 1024, + "end": 1280, + "bpb": 1.0363256199642377 + }, + { + "start": 1280, + "end": 1536, + "bpb": 1.0293503892901525 + }, + { + "start": 1536, + "end": 1792, + "bpb": 1.0270513457476949 + }, + { + "start": 1792, + "end": 2048, + "bpb": 1.0255301775348173 + }, + { + "start": 2048, + "end": 2304, + "bpb": 1.0171632048394768 + }, + { + "start": 2304, + "end": 2560, + "bpb": 1.0150940036004366 + }, + { + "start": 2560, + "end": 2816, + "bpb": 1.0132597834490384 + }, + { + "start": 2816, + "end": 3072, + "bpb": 1.0154303287433495 + }, + { + "start": 3072, + "end": 3328, + "bpb": 1.0125085420227948 + }, + { + "start": 3328, + "end": 3584, + "bpb": 1.0117220799135607 + }, + { + "start": 3584, + "end": 3840, + "bpb": 1.0093052242865306 + }, + { + "start": 3840, + "end": 4096, + "bpb": 1.0067290549863526 + }, + { + "start": 4096, + "end": 4352, + "bpb": 1.007627760298138 + }, + { + "start": 4352, + "end": 4608, + "bpb": 1.005475773581573 + }, + { + "start": 4608, + "end": 4864, + "bpb": 1.0011348016292028 + }, + { + "start": 4864, + "end": 5120, + "bpb": 1.0025369118565095 + }, + { + "start": 5120, + "end": 5376, + "bpb": 0.9978363303279962 + }, + { + "start": 5376, + "end": 5632, + "bpb": 0.9936016054109623 + }, + { + "start": 5632, + "end": 5888, + "bpb": 0.9930789448128515 + }, + { + "start": 5888, + "end": 6144, + "bpb": 0.988120986726172 + }, + { + "start": 6144, + "end": 6400, + "bpb": 0.9878455863729801 + }, + { + "start": 6400, + "end": 6656, + "bpb": 0.988322568304946 + }, + { + "start": 6656, + "end": 6912, + "bpb": 0.9895607000315525 + }, + { + "start": 6912, + "end": 7168, + "bpb": 0.9924135354945043 + }, + { + "start": 7168, + "end": 7424, + "bpb": 0.9887797597227126 + }, + { + "start": 7424, + "end": 7680, + "bpb": 0.9863644464805957 + }, + { + "start": 7680, + "end": 7936, + "bpb": 0.9852729515784135 + }, + { + "start": 7936, + "end": 8192, + "bpb": 0.9838855089135268 + } + ], + "val": 1.0127322889400117 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25-ctx8192/run.json b/experiments/think-d12-r11.25-ctx8192/run.json new file mode 100644 index 0000000000000000000000000000000000000000..c9dfef8e7cdc56887e670c7c0a925609d0ba770c --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/run.json @@ -0,0 +1,10 @@ +{ + "experiment_id": "think-d12-r11.25-ctx8192", + "stage": "base", + "base_experiment_id": "think-d12-r11.25-ctx8192", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "407a5074e0bf3730", + "wandb_run_id": "e3483a4b", + "created_at": 1783693257 +} diff --git a/experiments/think-d12-r11.25-ctx8192/summary.json b/experiments/think-d12-r11.25-ctx8192/summary.json new file mode 100644 index 0000000000000000000000000000000000000000..3a2267f78ba4f11440074975edb5854408a02768 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/summary.json @@ -0,0 +1,69 @@ +{ + "experiment_id": "think-d12-r11.25-ctx8192", + "stage": "base", + "base_experiment_id": "think-d12-r11.25-ctx8192", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "dataset": "jbduran/think-dataset", + "dataset_revision": "main", + "step": 2362, + "depth": 12, + "target_param_data_ratio": 11.25, + "training_tokens": 1238368256, + "final_sampled_val_bpb": 1.0395519592251896, + "minimum_sampled_val_bpb": 1.0395519592251896, + "full_val_bpb": 1.0127322889400117, + "core_metric": null, + "centered_results": null, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is not yet fully developed. The capital of the United States is not yet fully developed" + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is the symbol of the gold, and the symbol of the silver. The gold is" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be the day of the week. \n\nThe day of the week is the same as" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is the same as hot. \n\nThe hot is the same as hot. \n\nThe" + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, 2. The moon, 3. The" + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is the color of the skin of the face, and the color of the skin." + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the" + } + ], + "unconditioned_samples": [ + "<|bos|>ALIENS. A concern called industrial, in which all miners of ability for useful labor were engaged in obtaining industrial materials for making snuffers. Although it would be a rude and untenable enterprise to make different classes of miners dispose of goods for profit, different miners differing between the quality of the material misspelled and its quantity, it always staggers the mind with the idea of the matter which it concerns.\u00b9 Four or five shopkeepers are seen at so many tradeshops in town near together in towns and villages. \n\nMoney is better paid to supply the needs of skilled men than it is in overcrowded", + "<|bos|>37374.31 617.471.11 379.75 \n\nSwinburne, Jes. 10, 335.\n\nStatistical Index. \n\nCoates, erance, 1333. \n\nPurple Debenture, 1460.\n\nGenetic Index. \n\nSwinfenning, 51858.29, 1319. \n\nFree Presses. \n\nRidgens-Pantrepous, 25. \n\nSprayl-Power, regular exercise, 1700. \n\nPreachers and Teachers of the Schools, 401\u20132. \n\n", + "<|bos|>URE FOOD AND THE BODY' \n\nIn such a commonwealth Siamese readers might find in Kumber's Essays, or Mabon's Vol. of Tobit and St. Jerome's Lives, a sound, a clear and satisfactory explanation of this phrase. May Lady Cassius inform the reverend Society from which this paragraph is borrowed that in this hour of peril men and women may \"paint to him,\" and bewail Rest and Treatment. In both of these circles there are three main meanings attached to the phrase. One is, with the excessive reference in hotel-keepers, donkeys, or hares; another, with alligators", + "<|bos|>HENRY MART 140. \n\nLeblay, Mr. De Martyn's invention of music, 4. IX.\n\nTRANSLATOR'S NOTE.-Send forth a translation of the Notes which were received by me, translated from the Musical \n\nCommission's Calendar.\n\nHis performance we cannot altogether estimate, but St. Columba gave us confirmation of his inventions, 30. XXVIIii, 18. Eh, What (Georgics, I, 177); 'Slightest book that ever was written' (Sonn., lies 28\u00bd), a work of high merit read with honour,\n\nQu\u00e6rese", + "<|bos|>Harvard School, IV Department of Education, 1843-1972. \n\n2 Henry State League, LL. concerning Courses in Medicine and the Arts, pp. 22 et seq.\n\nHistory of the Monroe Doctrine, by one who has visited Europe, compiled from European Authorities.\n\nNew York: N. Y. \n\nExaminer, Vol. XXXI, \n\nApril, 1917, p. 88.\n\nPamphlet on Scien tific Methods of Education (\"Outlines of the Maladies and Defects of the Methods of Industrial Society,\" by Dr. Jevons). \n\nNew York, September and October, 19", + "<|bos|>The Bird reflects upon. his. \n\nThe Clipper. \n\nVapour.\n\nUntil within a few weeks the admission of the truth to our beloved Bird was fatal to that race, little cared for neither in her recorded history nor since they married, and still less as regards her character. Her reign ended; and when she died only after a few months good for nothing the country felt herself well restored to health. She began now to see her way. Our dear bird became as dear to her as the Christian mother; she began to see her way clearer to her senses; and as her thoughts turned, and freedom fell back, she", + "<|bos|>Army of the Cumberland and Arkansas Army, and an Army of the Potomac under the command of Martin Robertson.\n\nHEADQUARTERS CAMP THIRDQUARTERS, THIRD BRIG 1ST BRIG 1ST BRIG 1ST BRIG 1ST BRIG \n\n6 8 8 \n\nMCLQUERISHER'S STATION, 9 P.M. \n\nMY ARMY, CAL. \n\nEnlarged with orders by the War Department.\n\nHeadquarters Camp War Department, Clope Ridge, Va., September 15, 1864. \n\n6 P.M The Confederates tend S. M. Camp are in the Confederate service hospital at N. C.", + "<|bos|>the eightieth year of his age.\n\nI had never been conferring, like the palette and the gallens at which I used to sit. never had thought it wrong to give the faintest hint of this prudery, which I trust is always requested of the upholder at his housekeeping, as shall appear by the order and directions accompanying it. So while I was listening to the wise old voice of the tender bride calling alone in her measure that nonsense of patriarchal impiety. It was all the more gratifying when I heard that the Governor of Shetton is now apparently labouring in the same breath, when he speaks" + ], + "training_time_seconds": 9008.03685593605, + "stage_training_flops": 1.9399969190612828e+18, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1.9399969190612828e+18, + "config_fingerprint": "407a5074e0bf3730", + "git_commit_sha": "083cd7f99484b5e894a23e4a093de6d339c412ea", + "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/e3483a4b", + "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r11.25-ctx8192", + "dataset_fingerprint": "63a5e6be81591d82", + "tokenizer_fingerprint": "03c4f62e7a9d0c3b", + "unique_train_tokens": 0 +} diff --git a/experiments/think-d12-r11.25-ctx8192/tokenizer/experiment_tokenizer.json b/experiments/think-d12-r11.25-ctx8192/tokenizer/experiment_tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..8a28b84314d5995d3b64f34451ce3a694d901f80 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/tokenizer/experiment_tokenizer.json @@ -0,0 +1,18 @@ +{ + "experiment_id": "think-d12-r11.25-ctx8192", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 24, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "created_at": 1783708207 +} diff --git a/experiments/think-d12-r11.25-ctx8192/tokenizer/token_bytes.pt b/experiments/think-d12-r11.25-ctx8192/tokenizer/token_bytes.pt new file mode 100644 index 0000000000000000000000000000000000000000..01d1ec4aab9e8a7d205c3b3ffbeb8da0e9a62db1 --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/tokenizer/token_bytes.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1 +size 132649 diff --git a/experiments/think-d12-r11.25-ctx8192/tokenizer/tokenizer.pkl b/experiments/think-d12-r11.25-ctx8192/tokenizer/tokenizer.pkl new file mode 100644 index 0000000000000000000000000000000000000000..a17bd392980021628053b95d6425fc556aad527a --- /dev/null +++ b/experiments/think-d12-r11.25-ctx8192/tokenizer/tokenizer.pkl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1 +size 404071 diff --git a/experiments/think-d12-r11.25/base_checkpoints/meta_000500.json b/experiments/think-d12-r11.25/base_checkpoints/meta_000500.json new file mode 100644 index 0000000000000000000000000000000000000000..4346b0a0961c0461b449f5bf3e53e0e367f9f980 --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/meta_000500.json @@ -0,0 +1,61 @@ +{ + "step": 500, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": null + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 2, + "pos": 62184769, + "epoch": 1, + "pq_idx": 2, + "rg_idx": 62184769 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 3.5893779623775406, + "total_training_time": 1290.4532148838043 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/base_checkpoints/meta_001000.json b/experiments/think-d12-r11.25/base_checkpoints/meta_001000.json new file mode 100644 index 0000000000000000000000000000000000000000..d44aff60900e63f44ee57a352b66725c582f2c36 --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/meta_001000.json @@ -0,0 +1,61 @@ +{ + "step": 1000, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": null + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 5, + "pos": 24336769, + "epoch": 1, + "pq_idx": 5, + "rg_idx": 24336769 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 3.4419165825253133, + "total_training_time": 2609.577807664871 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/base_checkpoints/meta_001500.json b/experiments/think-d12-r11.25/base_checkpoints/meta_001500.json new file mode 100644 index 0000000000000000000000000000000000000000..1991d9750741add910d01954f8482d4e2247e360 --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/meta_001500.json @@ -0,0 +1,61 @@ +{ + "step": 1500, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": null + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 7, + "pos": 86488769, + "epoch": 1, + "pq_idx": 7, + "rg_idx": 86488769 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 3.2465426140603015, + "total_training_time": 3929.598204135895 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/base_checkpoints/meta_002000.json b/experiments/think-d12-r11.25/base_checkpoints/meta_002000.json new file mode 100644 index 0000000000000000000000000000000000000000..488a6f07b4fea050f2e89f34893aad59564826a0 --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/meta_002000.json @@ -0,0 +1,61 @@ +{ + "step": 2000, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": null + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 10, + "pos": 48640769, + "epoch": 1, + "pq_idx": 10, + "rg_idx": 48640769 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 3.2442641345309373, + "total_training_time": 5249.494728565216 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/base_checkpoints/meta_002362.json b/experiments/think-d12-r11.25/base_checkpoints/meta_002362.json new file mode 100644 index 0000000000000000000000000000000000000000..e175df81e356e808f571537ad3eb29ed517c3335 --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/meta_002362.json @@ -0,0 +1,61 @@ +{ + "step": 2362, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": null + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 12, + "pos": 38438817, + "epoch": 1, + "pq_idx": 12, + "rg_idx": 38438817 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 3.074421420856799, + "total_training_time": 6205.646646976471 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/base_checkpoints/model_000500.pt b/experiments/think-d12-r11.25/base_checkpoints/model_000500.pt new file mode 100644 index 0000000000000000000000000000000000000000..adb2ea235f50e379dec7aedafe47be61861bedab --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/model_000500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21fbdff366e5db3fa2f254b4b98c763cc70ba95722242fa032d8d21b956b694f +size 792761399 diff --git a/experiments/think-d12-r11.25/base_checkpoints/model_001000.pt b/experiments/think-d12-r11.25/base_checkpoints/model_001000.pt new file mode 100644 index 0000000000000000000000000000000000000000..6c50be693db2de57ed0fc9edeb4808310b2631ab --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/model_001000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6f0a862044e51b8cac250cf1e4f26af7f8347fd887b8c30af08addb879b6494d +size 792761399 diff --git a/experiments/think-d12-r11.25/base_checkpoints/model_001500.pt b/experiments/think-d12-r11.25/base_checkpoints/model_001500.pt new file mode 100644 index 0000000000000000000000000000000000000000..6c11f7b0d9217759f1212acf26a448949fffd5d8 --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/model_001500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:326bddc26c100de33310f9164dc873af6099a9b7760527a8707c256988a8ec7a +size 792761399 diff --git a/experiments/think-d12-r11.25/base_checkpoints/model_002000.pt b/experiments/think-d12-r11.25/base_checkpoints/model_002000.pt new file mode 100644 index 0000000000000000000000000000000000000000..4ec265330fa790437406e93a5448ad62eb39fe99 --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/model_002000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a5d101975722cffc43eb4f35aa967a5afd397b159da3521e5a1acbd3819d466e +size 792761399 diff --git a/experiments/think-d12-r11.25/base_checkpoints/model_002362.pt b/experiments/think-d12-r11.25/base_checkpoints/model_002362.pt new file mode 100644 index 0000000000000000000000000000000000000000..6cf06b066423745b22aa44519e4b864d5bdc27d6 --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/model_002362.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:04331c52ed8fa7259f350e4ec72d0dd6602451cfd75a5773a4c17ac5c141ea7e +size 792761399 diff --git a/experiments/think-d12-r11.25/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r11.25/base_checkpoints/optim_000500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..79c1822fb2dc78fb93423cdc8813dfc59fea2a94 --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/optim_000500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d39a131089a476a202ae932f21d9398b225d0b8e4147a58fbdff797914d34976 +size 1246165237 diff --git a/experiments/think-d12-r11.25/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r11.25/base_checkpoints/optim_001000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..f055caef01d26659f49ac7ae7f17083f45d82d4a --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/optim_001000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e4971566eb3d4aa5d5fe29b3a3b77f56581b61713e5c1d162debb8a409c02118 +size 1246165237 diff --git a/experiments/think-d12-r11.25/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r11.25/base_checkpoints/optim_001500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..275f823331fcb1cd614bc3d8e15f97265d98d85a --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/optim_001500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:05ce87d44bfd6e9b4d4c1e643a2a5aa4b169119913ccbdf3625d7cfa7313b342 +size 1246165237 diff --git a/experiments/think-d12-r11.25/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r11.25/base_checkpoints/optim_002000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..b3b5e1bd94301a192b14e234e17f4f28786b5538 --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/optim_002000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1b58f68fd887360ce99e8456aecf706b1bcf5c8210194721389698ec658237b1 +size 1246165237 diff --git a/experiments/think-d12-r11.25/base_checkpoints/optim_002362_rank0.pt b/experiments/think-d12-r11.25/base_checkpoints/optim_002362_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..47439d09d2d9170f2458f5ad529c461a8e8c7f80 --- /dev/null +++ b/experiments/think-d12-r11.25/base_checkpoints/optim_002362_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:09aa5632a4b33981a1b2fbd97254d0050ecb7916d0c242f1d195c0726be314d7 +size 1246165237 diff --git a/experiments/think-d12-r11.25/config.json b/experiments/think-d12-r11.25/config.json new file mode 100644 index 0000000000000000000000000000000000000000..3d25cc9589b19a727199d6a7f46b93beb73eff45 --- /dev/null +++ b/experiments/think-d12-r11.25/config.json @@ -0,0 +1,55 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r11.25", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 24, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r11.25", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11.25" + ] + }, + "config_fingerprint": "3b5a68714770b6af", + "artifact_path": "experiments/think-d12-r11.25" +} diff --git a/experiments/think-d12-r11.25/evals/samples.json b/experiments/think-d12-r11.25/evals/samples.json new file mode 100644 index 0000000000000000000000000000000000000000..577869f25a98c31214b820db1b56d340add64022 --- /dev/null +++ b/experiments/think-d12-r11.25/evals/samples.json @@ -0,0 +1,48 @@ +{ + "model": "base_model (step 2362)", + "step": 2362, + "bpb": {}, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is 10,000,000 francs, and the capital of the United" + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is the gold of the \n\nUnited States. It is the gold of the United States" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nThe day is Sunday, and the day is Sunday. \n\nThe" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is the opposite of cold. \n\nThe opposite of cold is the opposite of cold." + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, and the stars. \n\n2." + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is the color of the sky. \n\nThe color of the sky is a color of" + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times the number of times the number of times the number of times" + } + ], + "unconditioned_samples": [ + "<|bos|>Monthillahivan-this is worthy, they say, of the glory of God's presence, our Father being now to come again! \n\nDead. The priest who said Lord! so between thy guilty hands exulting and curse shoot; do thy honors sweetly, O Father! touch me not with panic, nor miss me with holy surprise, nor sigh (say) tost away my days, nor desolate them; thou wert sent by heaven, and seen with so much care, that Thou near thy sins didst repair them; how art Thou now to come thus early, and bear unto me so divinely? not", + "<|bos|>370 \n\nPaxton's Introduction to Education, or the true philosophy of the schools criticised, is differentiated. In Rossetti's plan, as already noted, it formulated in these brief articles, there should be but two different versions of the 73\n\n-flat elements; in Rossetti's, we commonly use the broad current of their general aim. That which has been called a positive and material-is aptly called a negative-was given with corresponding emphasis at three different times, at different epochs, but the THREE are frequently gods regular in origin, and preserved from the violence of adjustment and shift, rather than dormant and deliberate meaning of", + "<|bos|>ROrepoys and Willieot's book on the Siamese Law. \n\nA Letter from Frank Perrivsky to a Prince of Wales. By Kinsman Keith. \n\nTHE \n\nMAY, 1923. \n\nIn Paper \n\nWith 25 Illustrations. Parts. $2 $9 $14 6 $2.50 \n\nIN POLITICAL IDOLSES. \n\nAmerican Law. By George W. Resting, LL.D., Professor of Political Economy in Princeton University. \n\n$18 $2.50 \n\nSamson Minot, hotel-keeper. $3 $7\n\nTHE SCALE OF SIZE.", + "<|bos|>HENRY MARTYN SAXON, THE HINDUS CHRISTURIENT. \n\nEDWARD IRVING, of the College of the Anatomy School of \n\nDurham, Surrey.\n\nEDWARD PERCY BAKER, OF ALICE COLLEGE, whose personal appearance is now in print, was born at Stanneley, Surrey, July 18, 1794. He has been student in the Company's Military College at Woolwich for more than one year, for his learning and industry in his profession and studies. He has written a book entitled History, Economics and Political Science, which lies nearly at our very door, entitled History, Political Science and Political \n\nScience. It is", + "<|bos|> HOUSE OF THE ANGELS. \n\nFrom the City of St. Ann. \n\n2 vols. 3s. My Last in a Garden. I reserve for the fifth edition, in manuscript, a full account of ancient His tory. In \n\n1 vol. 5, a short history of Nero and Herod. from 6 to \n\n10 Years, both kept in this Library.\n\n4 \n\nLibrary of the Inducci EN QUANDRON. \n\nFRONTIER, SAMUEL, Dean of Carlisle, President of the\n\nAmerican Board of Works, 3 vols. 2 vols. 3s. 6d. \n\nLibrary", + "<|bos|>The Bird reflects the world, and swims the eagle's web.-Van Isle.\n\nSHOP'S \n\nREACH \n\nNEGARYELY GENIAL SHOP FINANCIERS RAIDER \n\nREAR HEADS,.} \n\nREPUBLICS, WHILE SHE UNDER FULLER'S PORT, \n\nREP Philosophers, that they be not \n\nPharaoh's patterns good for nothing, and valiant men for that which is nothing; \n\nCLY VAUS\u00d2 EXOVENT \u03b4\u1f72 \u03bf\u03cd\u03c3\u03b1\u03b9 \u0391\u03b4\u03af\u03b4\u03b5\u03b9ANTA\u03c1, \u1f15\u03b4\u03b1\u03c1\u03c9\u03bd \u03c3\u03bf\u03c5\u03b4\u03bf\u1f7a\u03c2 The Queen eschews", + "<|bos|>'Eau du Monarch beau ou H\u00f4tel de Voodjiches.\" (The same French Inspecteur :) \"Son jours\n\nA \u00e9t\u00e9 autant \u00e0 tomboi les m\u00eames Premi\u00e8res avec une princesse qui sont d'ombres le miraculeur bien de France le souffrir. Les c\u00f4tes de ceux qui le sont j\u00e9sibres.\" (The French Directors.)\n\neffected these river improvements in effeminate cases, in the The Executive has eschewed the corrupdemell\u00e8 hrs. Nieuw Ga", + "<|bos|>the eight (250) series.\n\nThe pressure is already great.\n\nashions like the palette de la \n\nLast chapter, however, it may be noted, and it is a most common practice in such a period for small and countless series to transform the palette de la up half at the sauce, and it is a matter of considerable importance to seeing the clothes-holders uniformly the marks that indicate the supply of the palette de la into that respectable width whence their sulky tints are generally returned. \n\nThe colour represents the price of the garment at the time apparently labouring in the work. Delivery is possible" + ] +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/evals/val_bpb.json b/experiments/think-d12-r11.25/evals/val_bpb.json new file mode 100644 index 0000000000000000000000000000000000000000..34f6714207d8b0ee7715eeda9acd0ca344802cf2 --- /dev/null +++ b/experiments/think-d12-r11.25/evals/val_bpb.json @@ -0,0 +1,54 @@ +{ + "model": "base_model (step 2362)", + "step": 2362, + "bpb": { + "val_per_position": [ + { + "start": 0, + "end": 256, + "bpb": 1.1255410237731462 + }, + { + "start": 256, + "end": 512, + "bpb": 1.0636677720059344 + }, + { + "start": 512, + "end": 768, + "bpb": 1.0506832286096848 + }, + { + "start": 768, + "end": 1024, + "bpb": 1.0456638664028364 + }, + { + "start": 1024, + "end": 1280, + "bpb": 1.038324560694976 + }, + { + "start": 1280, + "end": 1536, + "bpb": 1.0330914959872517 + }, + { + "start": 1536, + "end": 1792, + "bpb": 1.0308104068466784 + }, + { + "start": 1792, + "end": 2048, + "bpb": 1.027579795687179 + } + ], + "val": 1.0519195678472355 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean-1930s.json b/experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean-1930s.json new file mode 100644 index 0000000000000000000000000000000000000000..aa836c8edd73b612a1cc9c69196ef0398a23fd24 --- /dev/null +++ b/experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean-1930s.json @@ -0,0 +1,12 @@ +{ + "model": "base_model (step 2362)", + "step": 2362, + "bpb": { + "val": 1.0781689302026238 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json b/experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json new file mode 100644 index 0000000000000000000000000000000000000000..9fc110ec140db1d49acaca9fd71b69b47fcb85dd --- /dev/null +++ b/experiments/think-d12-r11.25/evals/val_bpb_on_think-dataset-clean.json @@ -0,0 +1,12 @@ +{ + "model": "base_model (step 2362)", + "step": 2362, + "bpb": { + "val": 1.081385374786877 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/run.json b/experiments/think-d12-r11.25/run.json new file mode 100644 index 0000000000000000000000000000000000000000..1004e46b9b3dd13da3a3c8aab534df801a665cc8 --- /dev/null +++ b/experiments/think-d12-r11.25/run.json @@ -0,0 +1,6 @@ +{ + "experiment_id": "think-d12-r11.25", + "stage": "base", + "wandb_run_id": null, + "migration_note": "Migrated from the pre-lineage repository layout." +} diff --git a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/meta_001065.json b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/meta_001065.json new file mode 100644 index 0000000000000000000000000000000000000000..2f52697c9b62b9b51836bb31924aadd0d871459c --- /dev/null +++ b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/meta_001065.json @@ -0,0 +1,38 @@ +{ + "step": 1065, + "val_bpb": 0.39271438585109175, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "model_tag": "d12", + "model_step": null, + "load_optimizer": 1, + "num_iterations": -1, + "max_seq_len": null, + "device_batch_size": 8, + "total_batch_size": null, + "embedding_lr": null, + "unembedding_lr": null, + "matrix_lr": null, + "init_lr_frac": 0.8, + "warmup_ratio": 0.0, + "warmdown_ratio": 0.5, + "final_lr_frac": 0.0, + "eval_every": -1, + "eval_tokens": 20971520, + "chatcore_every": -1, + "chatcore_max_cat": -1, + "chatcore_max_sample": 24, + "mmlu_epochs": 3, + "gsm8k_epochs": 4 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/model_001065.pt b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/model_001065.pt new file mode 100644 index 0000000000000000000000000000000000000000..637ffa7aa73462ac49b2503b5b34b31bfcd7b3ce --- /dev/null +++ b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/model_001065.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f1ef8886ea2cbd820baa9bc98368f99673efb1f5fdbd082196d573b291111bda +size 792761399 diff --git a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/optim_001065_rank0.pt b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/optim_001065_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..451dddf0789f3113d3b31c3792c2328b1f92da2c --- /dev/null +++ b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/checkpoints/optim_001065_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:db8c9df6e288618ed6362102e29edc3731267b4ff382154a55709b4d5a14f1d4 +size 1246165237 diff --git a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/config.json b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/config.json new file mode 100644 index 0000000000000000000000000000000000000000..cd6f5cd8b6a44a9a1d15b1422d203728a7bdfd83 --- /dev/null +++ b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/config.json @@ -0,0 +1,36 @@ +{ + "schema_version": 1, + "stage": "sft", + "experiment_id": "smoltalk-mmlu3-gsm8k4-v1", + "parent": { + "base_experiment_id": "think-d12-r11.25", + "checkpoint_step": 2362 + }, + "data": { + "recipe": "nanochat-default", + "mmlu_epochs": 3, + "gsm8k_epochs": 4 + }, + "training": { + "num_iterations": -1, + "device_batch_size": 8, + "eval_every": -1, + "chatcore_every": -1, + "save_every": 200 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": false, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "smoltalk-mmlu3-gsm8k4-v1", + "group": "think-d12-r11.25", + "tags": ["sft", "smoltalk", "mmlu3", "gsm8k4"] + }, + "historical": { + "lineage_inferred_from_checkpoint_metadata": true, + "wandb_was_disabled": true + } +} diff --git a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/run.json b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/run.json new file mode 100644 index 0000000000000000000000000000000000000000..4f172aadf1e6e2711e0b17ef0257df8bfdc9486e --- /dev/null +++ b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/run.json @@ -0,0 +1,6 @@ +{ + "experiment_id": "smoltalk-mmlu3-gsm8k4-v1", + "stage": "sft", + "wandb_run_id": null, + "migration_note": "Migrated from the pre-lineage repository layout." +} diff --git a/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/summary.json b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/summary.json new file mode 100644 index 0000000000000000000000000000000000000000..50e2e8ee27321ad7b1c85abc1f6f87bbf3704314 --- /dev/null +++ b/experiments/think-d12-r11.25/sft/smoltalk-mmlu3-gsm8k4-v1/summary.json @@ -0,0 +1,14 @@ +{ + "stage": "sft", + "experiment_id": "smoltalk-mmlu3-gsm8k4-v1", + "base_experiment_id": "think-d12-r11.25", + "parent_experiment_id": "think-d12-r11.25", + "parent_checkpoint_step": 2362, + "checkpoint_step": 1065, + "training_tokens": 558366720, + "flops_per_token": 887097900.0, + "stage_training_flops": 4.95325944741888e+17, + "inherited_parent_flops": 1.0985538793242624e+18, + "cumulative_pipeline_training_flops": 1.5938798240661504e+18, + "lineage_inferred_from_checkpoint_metadata": true +} diff --git a/experiments/think-d12-r11.25/summary.json b/experiments/think-d12-r11.25/summary.json new file mode 100644 index 0000000000000000000000000000000000000000..7da472fc36723c48d2b604c030abf02af4ee789c --- /dev/null +++ b/experiments/think-d12-r11.25/summary.json @@ -0,0 +1,69 @@ +{ + "experiment_id": "think-d12-r11.25", + "stage": "base", + "base_experiment_id": "think-d12-r11.25", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "dataset": "jbduran/think-dataset", + "dataset_revision": "main", + "step": 2362, + "depth": 12, + "target_param_data_ratio": 11.25, + "training_tokens": 1238368256, + "final_sampled_val_bpb": null, + "minimum_sampled_val_bpb": Infinity, + "full_val_bpb": 1.0519195678472355, + "core_metric": null, + "centered_results": null, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is 10,000,000 francs, and the capital of the United" + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is the gold of the \n\nUnited States. It is the gold of the United States" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nThe day is Sunday, and the day is Sunday. \n\nThe" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is the opposite of cold. \n\nThe opposite of cold is the opposite of cold." + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, the moon, and the stars. \n\n2." + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is the color of the sky. \n\nThe color of the sky is a color of" + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times the number of times the number of times the number of times" + } + ], + "unconditioned_samples": [ + "<|bos|>Monthillahivan-this is worthy, they say, of the glory of God's presence, our Father being now to come again! \n\nDead. The priest who said Lord! so between thy guilty hands exulting and curse shoot; do thy honors sweetly, O Father! touch me not with panic, nor miss me with holy surprise, nor sigh (say) tost away my days, nor desolate them; thou wert sent by heaven, and seen with so much care, that Thou near thy sins didst repair them; how art Thou now to come thus early, and bear unto me so divinely? not", + "<|bos|>370 \n\nPaxton's Introduction to Education, or the true philosophy of the schools criticised, is differentiated. In Rossetti's plan, as already noted, it formulated in these brief articles, there should be but two different versions of the 73\n\n-flat elements; in Rossetti's, we commonly use the broad current of their general aim. That which has been called a positive and material-is aptly called a negative-was given with corresponding emphasis at three different times, at different epochs, but the THREE are frequently gods regular in origin, and preserved from the violence of adjustment and shift, rather than dormant and deliberate meaning of", + "<|bos|>ROrepoys and Willieot's book on the Siamese Law. \n\nA Letter from Frank Perrivsky to a Prince of Wales. By Kinsman Keith. \n\nTHE \n\nMAY, 1923. \n\nIn Paper \n\nWith 25 Illustrations. Parts. $2 $9 $14 6 $2.50 \n\nIN POLITICAL IDOLSES. \n\nAmerican Law. By George W. Resting, LL.D., Professor of Political Economy in Princeton University. \n\n$18 $2.50 \n\nSamson Minot, hotel-keeper. $3 $7\n\nTHE SCALE OF SIZE.", + "<|bos|>HENRY MARTYN SAXON, THE HINDUS CHRISTURIENT. \n\nEDWARD IRVING, of the College of the Anatomy School of \n\nDurham, Surrey.\n\nEDWARD PERCY BAKER, OF ALICE COLLEGE, whose personal appearance is now in print, was born at Stanneley, Surrey, July 18, 1794. He has been student in the Company's Military College at Woolwich for more than one year, for his learning and industry in his profession and studies. He has written a book entitled History, Economics and Political Science, which lies nearly at our very door, entitled History, Political Science and Political \n\nScience. It is", + "<|bos|> HOUSE OF THE ANGELS. \n\nFrom the City of St. Ann. \n\n2 vols. 3s. My Last in a Garden. I reserve for the fifth edition, in manuscript, a full account of ancient His tory. In \n\n1 vol. 5, a short history of Nero and Herod. from 6 to \n\n10 Years, both kept in this Library.\n\n4 \n\nLibrary of the Inducci EN QUANDRON. \n\nFRONTIER, SAMUEL, Dean of Carlisle, President of the\n\nAmerican Board of Works, 3 vols. 2 vols. 3s. 6d. \n\nLibrary", + "<|bos|>The Bird reflects the world, and swims the eagle's web.-Van Isle.\n\nSHOP'S \n\nREACH \n\nNEGARYELY GENIAL SHOP FINANCIERS RAIDER \n\nREAR HEADS,.} \n\nREPUBLICS, WHILE SHE UNDER FULLER'S PORT, \n\nREP Philosophers, that they be not \n\nPharaoh's patterns good for nothing, and valiant men for that which is nothing; \n\nCLY VAUS\u00d2 EXOVENT \u03b4\u1f72 \u03bf\u03cd\u03c3\u03b1\u03b9 \u0391\u03b4\u03af\u03b4\u03b5\u03b9ANTA\u03c1, \u1f15\u03b4\u03b1\u03c1\u03c9\u03bd \u03c3\u03bf\u03c5\u03b4\u03bf\u1f7a\u03c2 The Queen eschews", + "<|bos|>'Eau du Monarch beau ou H\u00f4tel de Voodjiches.\" (The same French Inspecteur :) \"Son jours\n\nA \u00e9t\u00e9 autant \u00e0 tomboi les m\u00eames Premi\u00e8res avec une princesse qui sont d'ombres le miraculeur bien de France le souffrir. Les c\u00f4tes de ceux qui le sont j\u00e9sibres.\" (The French Directors.)\n\neffected these river improvements in effeminate cases, in the The Executive has eschewed the corrupdemell\u00e8 hrs. Nieuw Ga", + "<|bos|>the eight (250) series.\n\nThe pressure is already great.\n\nashions like the palette de la \n\nLast chapter, however, it may be noted, and it is a most common practice in such a period for small and countless series to transform the palette de la up half at the sauce, and it is a matter of considerable importance to seeing the clothes-holders uniformly the marks that indicate the supply of the palette de la into that respectable width whence their sulky tints are generally returned. \n\nThe colour represents the price of the garment at the time apparently labouring in the work. Delivery is possible" + ], + "training_time_seconds": 6205.646646976471, + "stage_training_flops": 1.0985538793242624e+18, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1.0985538793242624e+18, + "config_fingerprint": "3b5a68714770b6af", + "git_commit_sha": "205cddabbb34257a8a78cf63a47ae281e3b51ac5", + "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/None", + "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r11.25", + "dataset_fingerprint": "63a5e6be81591d82", + "tokenizer_fingerprint": "6e592b9b323f98bf", + "unique_train_tokens": 0 +} diff --git a/experiments/think-d12-r11.25/tokenizer/think_dataset_tokenizer.json b/experiments/think-d12-r11.25/tokenizer/think_dataset_tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..fea3141f8940ad53370916cd7a3dc8743edbf3ad --- /dev/null +++ b/experiments/think-d12-r11.25/tokenizer/think_dataset_tokenizer.json @@ -0,0 +1,20 @@ +{ + "dataset_repo": "jbduran/think-dataset", + "manifest_source_dataset": "institutional/institutional-books-1.0", + "num_train_shards": 24, + "expected_val_shard": "shard_00472.parquet", + "filters": { + "language": "eng", + "min_english_proportion": 0.9, + "year_max_exclusive": 1930, + "reject_invalid_date_types": true, + "ocr_min_inclusive": 90.0, + "ocr_disagreement_max_inclusive": 10.0, + "min_tokenizability": 95.0, + "min_tokens": 500, + "min_chars": 2000, + "min_pages": 3, + "min_sentences": 20, + "undated_rows_rejected": true + } +} \ No newline at end of file diff --git a/experiments/think-d12-r11.25/tokenizer/token_bytes.pt b/experiments/think-d12-r11.25/tokenizer/token_bytes.pt new file mode 100644 index 0000000000000000000000000000000000000000..01d1ec4aab9e8a7d205c3b3ffbeb8da0e9a62db1 --- /dev/null +++ b/experiments/think-d12-r11.25/tokenizer/token_bytes.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1 +size 132649 diff --git a/experiments/think-d12-r11.25/tokenizer/tokenizer.pkl b/experiments/think-d12-r11.25/tokenizer/tokenizer.pkl new file mode 100644 index 0000000000000000000000000000000000000000..a17bd392980021628053b95d6425fc556aad527a --- /dev/null +++ b/experiments/think-d12-r11.25/tokenizer/tokenizer.pkl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1 +size 404071 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_000500.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_000500.json new file mode 100644 index 0000000000000000000000000000000000000000..5c847924bceeb560949ff7f6cebd8cd50165b349 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_000500.json @@ -0,0 +1,137 @@ +{ + "step": 500, + "experiment_id": "think-d12-r20-2epoch", + "val_bpb": 1.3372388496121728, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-2epoch", + "wandb_run_id": "520f145b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,2epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 4995, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", + "experiment_id": "think-d12-r20-2epoch", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", + "tokenizer_fingerprint": "90b338f6bf263273", + "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-2epoch", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-2epoch", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 22, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "epochs": 2, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core-metric-max-per-task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-2epoch", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "2epoch" + ] + }, + "config_fingerprint": "23f3b18820c5ee2b", + "artifact_path": "experiments/think-d12-r20-2epoch" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-2epoch", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "23f3b18820c5ee2b" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 2, + "pos": 62184769, + "epoch": 1, + "pq_idx": 2, + "rg_idx": 62184769 + }, + "loop_state": { + "min_val_bpb": 1.3372388496121728, + "smooth_train_loss": 3.667125674624101, + "total_training_time": 1320.8929188251495, + "stage_training_flops": 232547388751872000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 232547388751872000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_001000.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_001000.json new file mode 100644 index 0000000000000000000000000000000000000000..d542a7ac0105130d8ccb3f2df7e1e8419a91022a --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_001000.json @@ -0,0 +1,137 @@ +{ + "step": 1000, + "experiment_id": "think-d12-r20-2epoch", + "val_bpb": 1.2543626382469548, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-2epoch", + "wandb_run_id": "520f145b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,2epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 4995, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", + "experiment_id": "think-d12-r20-2epoch", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", + "tokenizer_fingerprint": "90b338f6bf263273", + "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-2epoch", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-2epoch", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 22, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "epochs": 2, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core-metric-max-per-task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-2epoch", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "2epoch" + ] + }, + "config_fingerprint": "23f3b18820c5ee2b", + "artifact_path": "experiments/think-d12-r20-2epoch" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-2epoch", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "23f3b18820c5ee2b" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 5, + "pos": 24336769, + "epoch": 1, + "pq_idx": 5, + "rg_idx": 24336769 + }, + "loop_state": { + "min_val_bpb": 1.2543626382469548, + "smooth_train_loss": 3.4484858640243488, + "total_training_time": 2668.93452835083, + "stage_training_flops": 465094777503744000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 465094777503744000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_001500.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_001500.json new file mode 100644 index 0000000000000000000000000000000000000000..c6d5fa29bfcfba7daaa51791e245beb3a8f2f57e --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_001500.json @@ -0,0 +1,137 @@ +{ + "step": 1500, + "experiment_id": "think-d12-r20-2epoch", + "val_bpb": 1.2340784059707743, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-2epoch", + "wandb_run_id": "520f145b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,2epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 4995, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", + "experiment_id": "think-d12-r20-2epoch", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", + "tokenizer_fingerprint": "90b338f6bf263273", + "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-2epoch", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-2epoch", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 22, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "epochs": 2, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core-metric-max-per-task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-2epoch", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "2epoch" + ] + }, + "config_fingerprint": "23f3b18820c5ee2b", + "artifact_path": "experiments/think-d12-r20-2epoch" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-2epoch", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "23f3b18820c5ee2b" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 7, + "pos": 86488769, + "epoch": 1, + "pq_idx": 7, + "rg_idx": 86488769 + }, + "loop_state": { + "min_val_bpb": 1.2340784059707743, + "smooth_train_loss": 3.6208812604507203, + "total_training_time": 4014.0752940177917, + "stage_training_flops": 697642166255616000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 697642166255616000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_002000.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_002000.json new file mode 100644 index 0000000000000000000000000000000000000000..e5ce8769b531e47a67ba81a47fb6222c5dcc9c96 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_002000.json @@ -0,0 +1,137 @@ +{ + "step": 2000, + "experiment_id": "think-d12-r20-2epoch", + "val_bpb": 1.2034096822606946, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-2epoch", + "wandb_run_id": "520f145b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,2epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 4995, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", + "experiment_id": "think-d12-r20-2epoch", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", + "tokenizer_fingerprint": "90b338f6bf263273", + "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-2epoch", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-2epoch", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 22, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "epochs": 2, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core-metric-max-per-task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-2epoch", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "2epoch" + ] + }, + "config_fingerprint": "23f3b18820c5ee2b", + "artifact_path": "experiments/think-d12-r20-2epoch" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-2epoch", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "23f3b18820c5ee2b" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 10, + "pos": 48640769, + "epoch": 1, + "pq_idx": 10, + "rg_idx": 48640769 + }, + "loop_state": { + "min_val_bpb": 1.2034096822606946, + "smooth_train_loss": 3.433083085366257, + "total_training_time": 5361.500252485275, + "stage_training_flops": 930189555007488000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 930189555007488000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_002500.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_002500.json new file mode 100644 index 0000000000000000000000000000000000000000..ffc9a73ca481391c1d9f57d019eac3a72d81196b --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_002500.json @@ -0,0 +1,137 @@ +{ + "step": 2500, + "experiment_id": "think-d12-r20-2epoch", + "val_bpb": 1.1723409617937781, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-2epoch", + "wandb_run_id": "520f145b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,2epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 4995, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", + "experiment_id": "think-d12-r20-2epoch", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", + "tokenizer_fingerprint": "90b338f6bf263273", + "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-2epoch", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-2epoch", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 22, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "epochs": 2, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core-metric-max-per-task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-2epoch", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "2epoch" + ] + }, + "config_fingerprint": "23f3b18820c5ee2b", + "artifact_path": "experiments/think-d12-r20-2epoch" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-2epoch", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "23f3b18820c5ee2b" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 0, + "pos": 1487218, + "epoch": 2, + "pq_idx": 0, + "rg_idx": 1487218 + }, + "loop_state": { + "min_val_bpb": 1.1723409617937781, + "smooth_train_loss": 3.450269059013645, + "total_training_time": 6707.027107000351, + "stage_training_flops": 1162736943759360000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1162736943759360000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_003000.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_003000.json new file mode 100644 index 0000000000000000000000000000000000000000..51cf4e29f6ad883df2dbff79b353d81180f720c8 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_003000.json @@ -0,0 +1,137 @@ +{ + "step": 3000, + "experiment_id": "think-d12-r20-2epoch", + "val_bpb": 1.1416312127753048, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-2epoch", + "wandb_run_id": "520f145b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,2epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 4995, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", + "experiment_id": "think-d12-r20-2epoch", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", + "tokenizer_fingerprint": "90b338f6bf263273", + "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-2epoch", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-2epoch", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 22, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "epochs": 2, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core-metric-max-per-task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-2epoch", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "2epoch" + ] + }, + "config_fingerprint": "23f3b18820c5ee2b", + "artifact_path": "experiments/think-d12-r20-2epoch" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-2epoch", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "23f3b18820c5ee2b" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 2, + "pos": 63639218, + "epoch": 2, + "pq_idx": 2, + "rg_idx": 63639218 + }, + "loop_state": { + "min_val_bpb": 1.1416312127753048, + "smooth_train_loss": 3.17850348486724, + "total_training_time": 8053.040769100189, + "stage_training_flops": 1395284332511232000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1395284332511232000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_003500.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_003500.json new file mode 100644 index 0000000000000000000000000000000000000000..652173eeddc45eca75bc9f733eb1049ae68f7dc6 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_003500.json @@ -0,0 +1,137 @@ +{ + "step": 3500, + "experiment_id": "think-d12-r20-2epoch", + "val_bpb": 1.118717862479829, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-2epoch", + "wandb_run_id": "520f145b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,2epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 4995, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", + "experiment_id": "think-d12-r20-2epoch", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", + "tokenizer_fingerprint": "90b338f6bf263273", + "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-2epoch", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-2epoch", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 22, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "epochs": 2, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core-metric-max-per-task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-2epoch", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "2epoch" + ] + }, + "config_fingerprint": "23f3b18820c5ee2b", + "artifact_path": "experiments/think-d12-r20-2epoch" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-2epoch", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "23f3b18820c5ee2b" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 5, + "pos": 25791218, + "epoch": 2, + "pq_idx": 5, + "rg_idx": 25791218 + }, + "loop_state": { + "min_val_bpb": 1.118717862479829, + "smooth_train_loss": 3.0035985624321593, + "total_training_time": 9400.758259773254, + "stage_training_flops": 1627831721263104000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1627831721263104000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004000.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004000.json new file mode 100644 index 0000000000000000000000000000000000000000..e2d8e42f0a42f7323e12c546a8973aaff14ba5f2 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004000.json @@ -0,0 +1,137 @@ +{ + "step": 4000, + "experiment_id": "think-d12-r20-2epoch", + "val_bpb": 1.0983701045807681, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-2epoch", + "wandb_run_id": "520f145b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,2epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 4995, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", + "experiment_id": "think-d12-r20-2epoch", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", + "tokenizer_fingerprint": "90b338f6bf263273", + "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-2epoch", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-2epoch", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 22, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "epochs": 2, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core-metric-max-per-task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-2epoch", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "2epoch" + ] + }, + "config_fingerprint": "23f3b18820c5ee2b", + "artifact_path": "experiments/think-d12-r20-2epoch" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-2epoch", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "23f3b18820c5ee2b" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 7, + "pos": 87943218, + "epoch": 2, + "pq_idx": 7, + "rg_idx": 87943218 + }, + "loop_state": { + "min_val_bpb": 1.0983701045807681, + "smooth_train_loss": 3.169070686059073, + "total_training_time": 10746.176971197128, + "stage_training_flops": 1860379110014976000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1860379110014976000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004500.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004500.json new file mode 100644 index 0000000000000000000000000000000000000000..66ce6157e686363fa041bea5c0722b2fddb60580 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004500.json @@ -0,0 +1,137 @@ +{ + "step": 4500, + "experiment_id": "think-d12-r20-2epoch", + "val_bpb": 1.0766404457414214, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-2epoch", + "wandb_run_id": "520f145b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,2epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 4995, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", + "experiment_id": "think-d12-r20-2epoch", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", + "tokenizer_fingerprint": "90b338f6bf263273", + "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-2epoch", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-2epoch", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 22, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "epochs": 2, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core-metric-max-per-task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-2epoch", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "2epoch" + ] + }, + "config_fingerprint": "23f3b18820c5ee2b", + "artifact_path": "experiments/think-d12-r20-2epoch" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-2epoch", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "23f3b18820c5ee2b" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 10, + "pos": 50095218, + "epoch": 2, + "pq_idx": 10, + "rg_idx": 50095218 + }, + "loop_state": { + "min_val_bpb": 1.0766404457414214, + "smooth_train_loss": 2.9945010887296344, + "total_training_time": 12091.669941663742, + "stage_training_flops": 2092926498766848000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 2092926498766848000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004995.json b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004995.json new file mode 100644 index 0000000000000000000000000000000000000000..1d5c5a63c9a993fee6487d36ef8202a0eff1b594 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/meta_004995.json @@ -0,0 +1,137 @@ +{ + "step": 4995, + "experiment_id": "think-d12-r20-2epoch", + "val_bpb": 1.0624507774476128, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-2epoch", + "wandb_run_id": "520f145b", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,2epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 4995, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/base_checkpoints", + "experiment_id": "think-d12-r20-2epoch", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-2epoch/config.json", + "tokenizer_fingerprint": "90b338f6bf263273", + "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-2epoch", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-2epoch", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 22, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "epochs": 2, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core-metric-max-per-task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-2epoch", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "2epoch" + ] + }, + "config_fingerprint": "23f3b18820c5ee2b", + "artifact_path": "experiments/think-d12-r20-2epoch" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-2epoch", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "23f3b18820c5ee2b" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 0, + "pos": 320147, + "epoch": 3, + "pq_idx": 0, + "rg_idx": 320147 + }, + "loop_state": { + "min_val_bpb": 1.0624507774476128, + "smooth_train_loss": 2.90704286111097, + "total_training_time": 13423.92495751381, + "stage_training_flops": 2323148413631201280, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 2323148413631201280 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_000500.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_000500.pt new file mode 100644 index 0000000000000000000000000000000000000000..9408a39ed2f80cba1279c249f03b745875625265 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/model_000500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:08bb9da8ddaf2d78ebb0f6ba9103ca00e14d585b1afd6d0251b7424ab803c296 +size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_001000.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_001000.pt new file mode 100644 index 0000000000000000000000000000000000000000..1155eba52a00d58b1eedb1507a93afd5a567c3e2 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/model_001000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2c90c13ad23843d84d2bf5b14aa6983fb0180799994454e4b37ae7cb75bb334d +size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_001500.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_001500.pt new file mode 100644 index 0000000000000000000000000000000000000000..58c52e086c1f6351a9dc99e4b1da2a29a2600912 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/model_001500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1b0b4158ee911d17dd6fc2c6d0743995c6193fc9f141a88d4f108de16e115e46 +size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_002000.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_002000.pt new file mode 100644 index 0000000000000000000000000000000000000000..1c83943878ecec0704350c6175c8349da164feb8 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/model_002000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8bfca87221770ccc082e2f2a92d3841c36d99caecaf1210658a300129f31215b +size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_002500.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_002500.pt new file mode 100644 index 0000000000000000000000000000000000000000..f1b7d8887039f92750b95c6d7a94721e6deaa067 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/model_002500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bc24f494c1ef1a8b3f253abd5772cff42aeacfff5dfa69c01763b1ebcf79050e +size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_003000.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_003000.pt new file mode 100644 index 0000000000000000000000000000000000000000..fea89b5fbec67d7c85ef2985dc600a67c331069c --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/model_003000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d2860ed10f079d0fcbc95f4cf69bcdd5a07a81da9feb3106a5d976db4ba93661 +size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_003500.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_003500.pt new file mode 100644 index 0000000000000000000000000000000000000000..4a58e0fd9d8c70c7d416070bc6acd7a66c2a9fbc --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/model_003500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fbd68738a488021b353a396a70b875818376bdeecdff3d04fff8216fca80bdcb +size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_004000.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_004000.pt new file mode 100644 index 0000000000000000000000000000000000000000..2cb72bea035e3e8c3c48a777f34e6396cabfd5b8 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/model_004000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f132eea86799e79202ca6200a74a2b8805361f625032160594fea811e09298fd +size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_004500.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_004500.pt new file mode 100644 index 0000000000000000000000000000000000000000..31a6826c9e5dc3d42e1f26796ff9c888ab424ad4 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/model_004500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7f2d7eaf6cc11768364a4ff56a6c9059045893d1d45262dfc7151221f9e9cbb6 +size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/model_004995.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/model_004995.pt new file mode 100644 index 0000000000000000000000000000000000000000..5135e092880134bea1debd1a1eabd06b7a157624 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/model_004995.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c022d1ba93c3884b7ea7a5a057c60795804eb31d05cac61d836a782516eeb134 +size 792761690 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_000500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..92dbcccdef01bb5abbdf9b39146152645882f103 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_000500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2332fc532e05e4d2533a0894e1a5c756253970da580d7fb07346d00d9a55ddab +size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_001000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..cbcb8925fa8c1b8a4e12dfe01f32f37e09594a8e --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_001000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a546f0abf80972d9beda2221729f73e466f60393d37a6a6c5d9b7cdd60bba3b5 +size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_001500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..d8aff4671dc04dbd516a0925ab42dfe678b225d1 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_001500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b365410fa3b9f9eaa0c27af68055aadbebbf744ecb40b2544da19304e5882cda +size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_002000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..ec5e1c7d826f62948b9bd4941d81c49095e176a8 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_002000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:877e0f396df383cb9d9eb606759de51bdaa492a0d3555b20fff7ad147664e87a +size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_002500_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_002500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..e3068198ce4068101d1b3391a079f3a7f87e314c --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_002500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ce7df5dff6218eaf94685e547251c15cb8cda948d42b12c411ada56e8907d944 +size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_003000_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_003000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..ebf5b52b0a6d55b1f5a625503cec2060276bcd7c --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_003000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d53a7bd9273fc9b2c99be5318a1cab115821a82297b554835d666f4692c7599a +size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_003500_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_003500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..d70e72e7c1a7dd8a8da39fcd413c78e8c4cec2ad --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_003500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3e072400b0c09eeb8a00013d3f8736b0699095a8e9ddcf3f2a7cd9f918102dc8 +size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004000_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..2378b741f417de7b607ab6980b706a16a8e657b6 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:276f0d9e52c9854fc516b403d3a171a3ac69b52dd885968b2953312f8e3f7a5b +size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004500_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..ad303ec33098e173f9f583324826f6a83fb03e83 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b498b7f2e96dec7de1a2910c2677001d22102a87edbfb35f4f18d852bdeb41d8 +size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004995_rank0.pt b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004995_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..8b03797d50a7e74193a0be67177052024b09516e --- /dev/null +++ b/experiments/think-d12-r20-2epoch/base_checkpoints/optim_004995_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:289b156bc47fff9efd4b34a7fbb750ce5e27e58ce10f9d0945ad2d4850e41eb3 +size 1246165357 diff --git a/experiments/think-d12-r20-2epoch/config.json b/experiments/think-d12-r20-2epoch/config.json new file mode 100644 index 0000000000000000000000000000000000000000..38496e2e4425511b765de1c223404ef12a957f2b --- /dev/null +++ b/experiments/think-d12-r20-2epoch/config.json @@ -0,0 +1,56 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-2epoch", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 22, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "epochs": 2, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core-metric-max-per-task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-2epoch", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "2epoch" + ] + }, + "config_fingerprint": "23f3b18820c5ee2b", + "artifact_path": "experiments/think-d12-r20-2epoch" +} diff --git a/experiments/think-d12-r20-2epoch/evals/core.json b/experiments/think-d12-r20-2epoch/evals/core.json new file mode 100644 index 0000000000000000000000000000000000000000..63669f7323ced0eb6673a33d01485e0645ee48ba --- /dev/null +++ b/experiments/think-d12-r20-2epoch/evals/core.json @@ -0,0 +1,56 @@ +{ + "model": "base_model (step 4995)", + "step": 4995, + "bpb": {}, + "core_metric": 0.08720366535348274, + "core_results": { + "hellaswag_zeroshot": 0.2848038077354431, + "jeopardy": 0.0004723665479104966, + "bigbench_qa_wikidata": 0.12346833199262619, + "arc_easy": 0.32281145453453064, + "arc_challenge": 0.20904436707496643, + "copa": 0.5600000023841858, + "commonsense_qa": 0.3054873049259186, + "piqa": 0.547878086566925, + "openbook_qa": 0.24800001084804535, + "lambada_openai": 0.24063651263713837, + "hellaswag": 0.28659629821777344, + "winograd": 0.5677655935287476, + "winogrande": 0.5011838674545288, + "bigbench_dyck_languages": 0.10900000482797623, + "agi_eval_lsat_ar": 0.27391302585601807, + "bigbench_cs_algorithms": 0.4386363625526428, + "bigbench_operators": 0.06190476566553116, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.02866603620350361, + "coqa": 0.062132030725479126, + "boolq": 0.6067278385162354, + "bigbench_language_identification": 0.2505999803543091 + }, + "centered_results": { + "hellaswag_zeroshot": 0.04640507698059082, + "jeopardy": 0.0004723665479104966, + "bigbench_qa_wikidata": 0.12346833199262619, + "arc_easy": 0.09708193937937419, + "arc_challenge": -0.054607510566711426, + "copa": 0.12000000476837158, + "commonsense_qa": 0.1318591311573982, + "piqa": 0.0957561731338501, + "openbook_qa": -0.002666652202606201, + "lambada_openai": 0.24063651263713837, + "hellaswag": 0.048795064290364586, + "winograd": 0.13553118705749512, + "winogrande": 0.002367734909057617, + "bigbench_dyck_languages": 0.10900000482797623, + "agi_eval_lsat_ar": 0.09239128232002257, + "bigbench_cs_algorithms": 0.4386363625526428, + "bigbench_operators": 0.06190476566553116, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.02866603620350361, + "coqa": 0.062132030725479126, + "boolq": -0.03492674074674906, + "bigbench_language_identification": 0.17557753614335433 + }, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/evals/samples.json b/experiments/think-d12-r20-2epoch/evals/samples.json new file mode 100644 index 0000000000000000000000000000000000000000..1a577941b0f1680c511cbf0cfc24d1a73b93ea77 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/evals/samples.json @@ -0,0 +1,48 @@ +{ + "model": "base_model (step 4995)", + "step": 4995, + "bpb": {}, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is the capital of the kingdom of France, and the capital of the kingdom of France" + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is the gold of the sun, and the gold of the moon is the gold of" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be Sunday. \n\nThe day of the Lord's resurrection is the day of the Lord" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is the heat of the sun, and the heat of the sun is the heat of" + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, which is the centre of the earth's orbit" + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is the red, and the white, and the blue, and the yellow, and" + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the" + } + ], + "unconditioned_samples": [ + "<|bos|>auberstaden a topaz sick list they produce.\n\nREALISM \n\nWITH \n\nENGLAND 1666\n\n instructive SKELETON BURIPLOMATIC CORRESPONDENCE \n\nRepassed two eldest daughters belonging to one of the father's ex-officials in the same family, who died at an advanced age without having powerfully cultivated the talent necessary for their maintenance, These letters prove (were the value of our correspondence greater) how much encreased the readiness and ability of our soldiers and sailors, at this early age, to study mischief and plunder, and the necessity of effort to circumvent temptation, for we exercise discipline on certain beginnings of adventure which are not", + "<|bos|> Society chiefly composed of the most generous, disinterested, unchanging and neutral, one seeming to delight in the social advancement smoothly deferred, and another to endure that it shall in no case be free from thie drudgery which has before been tainted by corruption or illegality. ture.\n\nNeither of these views has been fully arrived at had the party been constituted wholly devoted to a particular church creed, according to the actual ideas which have been prevalent in at least one quarter of the country; and the design of the bill democracy has been to enrode a sort of '-lineal faction, with every attribute of a consurgent peer", + "<|bos|>wench, Morh'ot's Memid' vow, 102 Harright's Lane of London | Phila. \n\nEsopus Britann'. man are to his names, 37 the mistress of the temple, 140 of a . reconciliation beto all yet people, 26 of Ogil's wall, 27 of the site now in Heselt' pax 'mancient house', pet PLOT, AND, full blown, 45 house, 45 of Ruple, Angel's Island, No. 5, 44 \n\n36 said to happe, 46 to graci and '", + "<|bos|> INTERESTS OF BOYS. FOR YOUNG MEN BY A. B. MENDENHALL COFFEEWOOD, EDITOR OF JENNERS \n\n12353-58-1861.\n\nProfessor's\n\nCONTENTS A. B. MENDENHALL COFFEEWOOD'S MANUAL GYNNER SONS.. \n\n11 \n\nIntroduction \n\nCHARACTER \n\n12 \n\n\u2026\u2026\u2026... 12 \n\nINTRODUCTION 13 \n\nPRELIMINARY . \n\nTHE SOCIALIST NINETEENTH CENTURY A. B. MENDENHALL COFFEEWOOD'S MANUAL. \n\nInterest in the SOCIALIST MOVEMENT OF 1863 Twentieth Century EVIDENCE TRAFFIC SOCIALIST FEUDS ITS WILL Organized Social Democracy Conditions Gover", + "<|bos|>ma\n\nThey are good to see him, for all we have to drink on a plantation.\n\nSermon from Olivella in \"Dunmore's Soldier,\" \n\nTo the Servians of Richmond. \n\n(Exilles : one year.-LIFE and DEATH, 25 to 30.)\n\nHampton: January 25, 1776-7 \n\nBLOODY morning Our Thomas eat, nourish us with drink \n\nAnd to quench thirst we send him Seven days with Good\n\n bread and which he hath not drunk, we conceive it a very seasonable decertime to him, Which we may drink in with him on a", + "<|bos|>medyal Be. Jan. Sept. Feb. Esterhen Gebr.' (by Gesen.) \n\nMrs. he has blessed Be. day.' Anch., Seventy1; so is he was on shore.\n\nCHAP. XXVIII. \n\nAge as ordinarily blest;-the heavenly Monarchy; an There be gifted of the God of angels! It comes to the consciousness of Divine prerogative that this utterance enables Milton to encourage his inspirers to plunge into the vortex of this world of doubt and fears; to learn from it others may become irksome as he shrinks into himself, without The highest possible improvement afterwards", + "<|bos|>for Freeman. \n\n61. Meintum\n\nSmall, Articles called on the title and Acts again.\n\nActs by the Acts of 7th\n\nCont 1861-1918. \n\nApoplexies. Mystery was a much discussed matter in most days of munificence, and the request for compulsion among officials was leading.\n\nLetters Mat. xxii., xxvii., xxviii., xxxii., xxxiii., xxxiii., as well as ae letters in the Edward treaty of alliance were not called for in the The Iron Crusader, but the Mount Vernon consular offer of amity which these Nath. xxii", + "<|bos|> followed steamboats common to the three countries, this country and Russia, with the Metropolitan Districts of Manitoba, Galway, Ulster, and Connaught. The whole province is possessed of three great magazines for small ships, which are named the Minecigs. St Joseph at Prospect Hill, Kanssens at Harecastle, Stoketa and the Orinoco the other governors receiving the government of the province. \n\nThe principal city of the province is Noul. Every thing went on very comfortably till this great war which we call the present war, although our national disasters rose, owing to our" + ] +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/evals/val_bpb.json b/experiments/think-d12-r20-2epoch/evals/val_bpb.json new file mode 100644 index 0000000000000000000000000000000000000000..0779b3e88b1fdeb6f5749b00c704cf45fa7e76b3 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/evals/val_bpb.json @@ -0,0 +1,12 @@ +{ + "model": "base_model (step 4995)", + "step": 4995, + "bpb": { + "val": 1.0107521146719778 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/think-d12-r20-2epoch/run.json b/experiments/think-d12-r20-2epoch/run.json new file mode 100644 index 0000000000000000000000000000000000000000..3285685ff027cf001eb6ed8b8ff499bae16aea77 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/run.json @@ -0,0 +1,10 @@ +{ + "experiment_id": "think-d12-r20-2epoch", + "stage": "base", + "base_experiment_id": "think-d12-r20-2epoch", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "95d46d80121d6088", + "wandb_run_id": "520f145b", + "created_at": 1781532282 +} diff --git a/experiments/think-d12-r20-2epoch/summary.json b/experiments/think-d12-r20-2epoch/summary.json new file mode 100644 index 0000000000000000000000000000000000000000..7ff5139d1291229f45175d4e01694851eed495c4 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/summary.json @@ -0,0 +1,93 @@ +{ + "experiment_id": "think-d12-r20-2epoch", + "stage": "base", + "base_experiment_id": "think-d12-r20-2epoch", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "dataset": "jbduran/think-dataset", + "dataset_revision": "main", + "step": 4995, + "depth": 12, + "target_param_data_ratio": null, + "training_tokens": 2618818560, + "final_sampled_val_bpb": 1.0624507774476128, + "minimum_sampled_val_bpb": 1.0624507774476128, + "full_val_bpb": 1.0107521146719778, + "core_metric": 0.08720366535348274, + "centered_results": { + "hellaswag_zeroshot": 0.04640507698059082, + "jeopardy": 0.0004723665479104966, + "bigbench_qa_wikidata": 0.12346833199262619, + "arc_easy": 0.09708193937937419, + "arc_challenge": -0.054607510566711426, + "copa": 0.12000000476837158, + "commonsense_qa": 0.1318591311573982, + "piqa": 0.0957561731338501, + "openbook_qa": -0.002666652202606201, + "lambada_openai": 0.24063651263713837, + "hellaswag": 0.048795064290364586, + "winograd": 0.13553118705749512, + "winogrande": 0.002367734909057617, + "bigbench_dyck_languages": 0.10900000482797623, + "agi_eval_lsat_ar": 0.09239128232002257, + "bigbench_cs_algorithms": 0.4386363625526428, + "bigbench_operators": 0.06190476566553116, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.02866603620350361, + "coqa": 0.062132030725479126, + "boolq": -0.03492674074674906, + "bigbench_language_identification": 0.17557753614335433 + }, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is the capital of the kingdom of France, and the capital of the kingdom of France" + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is the gold of the sun, and the gold of the moon is the gold of" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be Sunday. \n\nThe day of the Lord's resurrection is the day of the Lord" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is the heat of the sun, and the heat of the sun is the heat of" + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, which is the centre of the earth's orbit" + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is the red, and the white, and the blue, and the yellow, and" + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the number of the number of the number of the number of the" + } + ], + "unconditioned_samples": [ + "<|bos|>auberstaden a topaz sick list they produce.\n\nREALISM \n\nWITH \n\nENGLAND 1666\n\n instructive SKELETON BURIPLOMATIC CORRESPONDENCE \n\nRepassed two eldest daughters belonging to one of the father's ex-officials in the same family, who died at an advanced age without having powerfully cultivated the talent necessary for their maintenance, These letters prove (were the value of our correspondence greater) how much encreased the readiness and ability of our soldiers and sailors, at this early age, to study mischief and plunder, and the necessity of effort to circumvent temptation, for we exercise discipline on certain beginnings of adventure which are not", + "<|bos|> Society chiefly composed of the most generous, disinterested, unchanging and neutral, one seeming to delight in the social advancement smoothly deferred, and another to endure that it shall in no case be free from thie drudgery which has before been tainted by corruption or illegality. ture.\n\nNeither of these views has been fully arrived at had the party been constituted wholly devoted to a particular church creed, according to the actual ideas which have been prevalent in at least one quarter of the country; and the design of the bill democracy has been to enrode a sort of '-lineal faction, with every attribute of a consurgent peer", + "<|bos|>wench, Morh'ot's Memid' vow, 102 Harright's Lane of London | Phila. \n\nEsopus Britann'. man are to his names, 37 the mistress of the temple, 140 of a . reconciliation beto all yet people, 26 of Ogil's wall, 27 of the site now in Heselt' pax 'mancient house', pet PLOT, AND, full blown, 45 house, 45 of Ruple, Angel's Island, No. 5, 44 \n\n36 said to happe, 46 to graci and '", + "<|bos|> INTERESTS OF BOYS. FOR YOUNG MEN BY A. B. MENDENHALL COFFEEWOOD, EDITOR OF JENNERS \n\n12353-58-1861.\n\nProfessor's\n\nCONTENTS A. B. MENDENHALL COFFEEWOOD'S MANUAL GYNNER SONS.. \n\n11 \n\nIntroduction \n\nCHARACTER \n\n12 \n\n\u2026\u2026\u2026... 12 \n\nINTRODUCTION 13 \n\nPRELIMINARY . \n\nTHE SOCIALIST NINETEENTH CENTURY A. B. MENDENHALL COFFEEWOOD'S MANUAL. \n\nInterest in the SOCIALIST MOVEMENT OF 1863 Twentieth Century EVIDENCE TRAFFIC SOCIALIST FEUDS ITS WILL Organized Social Democracy Conditions Gover", + "<|bos|>ma\n\nThey are good to see him, for all we have to drink on a plantation.\n\nSermon from Olivella in \"Dunmore's Soldier,\" \n\nTo the Servians of Richmond. \n\n(Exilles : one year.-LIFE and DEATH, 25 to 30.)\n\nHampton: January 25, 1776-7 \n\nBLOODY morning Our Thomas eat, nourish us with drink \n\nAnd to quench thirst we send him Seven days with Good\n\n bread and which he hath not drunk, we conceive it a very seasonable decertime to him, Which we may drink in with him on a", + "<|bos|>medyal Be. Jan. Sept. Feb. Esterhen Gebr.' (by Gesen.) \n\nMrs. he has blessed Be. day.' Anch., Seventy1; so is he was on shore.\n\nCHAP. XXVIII. \n\nAge as ordinarily blest;-the heavenly Monarchy; an There be gifted of the God of angels! It comes to the consciousness of Divine prerogative that this utterance enables Milton to encourage his inspirers to plunge into the vortex of this world of doubt and fears; to learn from it others may become irksome as he shrinks into himself, without The highest possible improvement afterwards", + "<|bos|>for Freeman. \n\n61. Meintum\n\nSmall, Articles called on the title and Acts again.\n\nActs by the Acts of 7th\n\nCont 1861-1918. \n\nApoplexies. Mystery was a much discussed matter in most days of munificence, and the request for compulsion among officials was leading.\n\nLetters Mat. xxii., xxvii., xxviii., xxxii., xxxiii., xxxiii., as well as ae letters in the Edward treaty of alliance were not called for in the The Iron Crusader, but the Mount Vernon consular offer of amity which these Nath. xxii", + "<|bos|> followed steamboats common to the three countries, this country and Russia, with the Metropolitan Districts of Manitoba, Galway, Ulster, and Connaught. The whole province is possessed of three great magazines for small ships, which are named the Minecigs. St Joseph at Prospect Hill, Kanssens at Harecastle, Stoketa and the Orinoco the other governors receiving the government of the province. \n\nThe principal city of the province is Noul. Every thing went on very comfortably till this great war which we call the present war, although our national disasters rose, owing to our" + ], + "training_time_seconds": 13423.92495751381, + "stage_training_flops": 2.3231484136312013e+18, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 2.3231484136312013e+18, + "config_fingerprint": "23f3b18820c5ee2b", + "git_commit_sha": "be028b0d222bac974ce9fa7525a78a6ed8d48dcb", + "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/520f145b", + "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r20-2epoch", + "dataset_fingerprint": "f743a21ad2774428", + "tokenizer_fingerprint": "90b338f6bf263273", + "unique_train_tokens": 1309305551, + "effective_epochs": 2.0001584488814252 +} diff --git a/experiments/think-d12-r20-2epoch/tokenizer/experiment_tokenizer.json b/experiments/think-d12-r20-2epoch/tokenizer/experiment_tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..839cfa5fffe02d709924f344017ef419dae7b3ab --- /dev/null +++ b/experiments/think-d12-r20-2epoch/tokenizer/experiment_tokenizer.json @@ -0,0 +1,18 @@ +{ + "experiment_id": "think-d12-r20-2epoch", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 22, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "created_at": 1781534280 +} diff --git a/experiments/think-d12-r20-2epoch/tokenizer/token_bytes.pt b/experiments/think-d12-r20-2epoch/tokenizer/token_bytes.pt new file mode 100644 index 0000000000000000000000000000000000000000..57df1c9e6f21a9cbfcb1131d0a730155c6eea7e1 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/tokenizer/token_bytes.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dde6733ef165ec5a1c250285a2f8bf7cf554039b0c23cbe98de70056c30d7901 +size 132649 diff --git a/experiments/think-d12-r20-2epoch/tokenizer/tokenizer.pkl b/experiments/think-d12-r20-2epoch/tokenizer/tokenizer.pkl new file mode 100644 index 0000000000000000000000000000000000000000..bebc9e00ea38cd78af03294576437cf234f34383 --- /dev/null +++ b/experiments/think-d12-r20-2epoch/tokenizer/tokenizer.pkl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:edf4a58d6a4778da458e66a7e10b61ba23d78c27404807aa4fb5e057a0ec964d +size 404047 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_000500.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_000500.json new file mode 100644 index 0000000000000000000000000000000000000000..c518526b52cefee7b83076a11ef97683fdaa6350 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/meta_000500.json @@ -0,0 +1,138 @@ +{ + "step": 500, + "experiment_id": "think-d12-r20-alt44", + "val_bpb": 1.3434480934243915, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-alt44", + "wandb_run_id": "4b198ee6", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,alt44", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", + "experiment_id": "think-d12-r20-alt44", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", + "tokenizer_fingerprint": "3dc1a109161e9100", + "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-alt44", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-alt44", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "min_train_shard": 44, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-alt44", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "alt44" + ] + }, + "config_fingerprint": "60b9890a3df95e25", + "artifact_path": "experiments/think-d12-r20-alt44" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-alt44", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "60b9890a3df95e25" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 2, + "pos": 62184769, + "epoch": 1, + "pq_idx": 2, + "rg_idx": 62184769 + }, + "loop_state": { + "min_val_bpb": 1.3434480934243915, + "smooth_train_loss": 3.6726269837043755, + "total_training_time": 1307.7268741130829, + "stage_training_flops": 232547388751872000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 232547388751872000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_001000.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_001000.json new file mode 100644 index 0000000000000000000000000000000000000000..af0705706a66e98c3b5e72c0d71b3316056c4a44 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/meta_001000.json @@ -0,0 +1,138 @@ +{ + "step": 1000, + "experiment_id": "think-d12-r20-alt44", + "val_bpb": 1.2710118912993729, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-alt44", + "wandb_run_id": "4b198ee6", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,alt44", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", + "experiment_id": "think-d12-r20-alt44", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", + "tokenizer_fingerprint": "3dc1a109161e9100", + "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-alt44", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-alt44", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "min_train_shard": 44, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-alt44", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "alt44" + ] + }, + "config_fingerprint": "60b9890a3df95e25", + "artifact_path": "experiments/think-d12-r20-alt44" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-alt44", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "60b9890a3df95e25" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 5, + "pos": 24336769, + "epoch": 1, + "pq_idx": 5, + "rg_idx": 24336769 + }, + "loop_state": { + "min_val_bpb": 1.2710118912993729, + "smooth_train_loss": 3.5156419568827517, + "total_training_time": 2644.638623714447, + "stage_training_flops": 465094777503744000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 465094777503744000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_001500.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_001500.json new file mode 100644 index 0000000000000000000000000000000000000000..e2456b76dc3c9a6a2913dd6c6342a09e4be5a783 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/meta_001500.json @@ -0,0 +1,138 @@ +{ + "step": 1500, + "experiment_id": "think-d12-r20-alt44", + "val_bpb": 1.2377739712547688, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-alt44", + "wandb_run_id": "4b198ee6", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,alt44", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", + "experiment_id": "think-d12-r20-alt44", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", + "tokenizer_fingerprint": "3dc1a109161e9100", + "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-alt44", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-alt44", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "min_train_shard": 44, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-alt44", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "alt44" + ] + }, + "config_fingerprint": "60b9890a3df95e25", + "artifact_path": "experiments/think-d12-r20-alt44" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-alt44", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "60b9890a3df95e25" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 7, + "pos": 86488769, + "epoch": 1, + "pq_idx": 7, + "rg_idx": 86488769 + }, + "loop_state": { + "min_val_bpb": 1.2377739712547688, + "smooth_train_loss": 3.3787051177159433, + "total_training_time": 3981.4400374889374, + "stage_training_flops": 697642166255616000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 697642166255616000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_002000.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_002000.json new file mode 100644 index 0000000000000000000000000000000000000000..9ca4283cbd9091b3bcd3e7c1f3ebf0eec8440dea --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/meta_002000.json @@ -0,0 +1,138 @@ +{ + "step": 2000, + "experiment_id": "think-d12-r20-alt44", + "val_bpb": 1.192725198125875, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-alt44", + "wandb_run_id": "4b198ee6", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,alt44", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", + "experiment_id": "think-d12-r20-alt44", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", + "tokenizer_fingerprint": "3dc1a109161e9100", + "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-alt44", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-alt44", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "min_train_shard": 44, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-alt44", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "alt44" + ] + }, + "config_fingerprint": "60b9890a3df95e25", + "artifact_path": "experiments/think-d12-r20-alt44" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-alt44", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "60b9890a3df95e25" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 10, + "pos": 48640769, + "epoch": 1, + "pq_idx": 10, + "rg_idx": 48640769 + }, + "loop_state": { + "min_val_bpb": 1.192725198125875, + "smooth_train_loss": 3.3640745998451362, + "total_training_time": 5317.553530216217, + "stage_training_flops": 930189555007488000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 930189555007488000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_002500.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_002500.json new file mode 100644 index 0000000000000000000000000000000000000000..d3234eb93f0269c6c91baec2205d57c940219265 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/meta_002500.json @@ -0,0 +1,138 @@ +{ + "step": 2500, + "experiment_id": "think-d12-r20-alt44", + "val_bpb": 1.1596344082826722, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-alt44", + "wandb_run_id": "4b198ee6", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,alt44", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", + "experiment_id": "think-d12-r20-alt44", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", + "tokenizer_fingerprint": "3dc1a109161e9100", + "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-alt44", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-alt44", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "min_train_shard": 44, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-alt44", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "alt44" + ] + }, + "config_fingerprint": "60b9890a3df95e25", + "artifact_path": "experiments/think-d12-r20-alt44" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-alt44", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "60b9890a3df95e25" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 13, + "pos": 10792769, + "epoch": 1, + "pq_idx": 13, + "rg_idx": 10792769 + }, + "loop_state": { + "min_val_bpb": 1.1596344082826722, + "smooth_train_loss": 3.2646331317328183, + "total_training_time": 6654.347955942154, + "stage_training_flops": 1162736943759360000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1162736943759360000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_003000.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_003000.json new file mode 100644 index 0000000000000000000000000000000000000000..aa445d6b55352e691fd2207c0e80c1d9e027ab3c --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/meta_003000.json @@ -0,0 +1,138 @@ +{ + "step": 3000, + "experiment_id": "think-d12-r20-alt44", + "val_bpb": 1.1237689589285966, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-alt44", + "wandb_run_id": "4b198ee6", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,alt44", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", + "experiment_id": "think-d12-r20-alt44", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", + "tokenizer_fingerprint": "3dc1a109161e9100", + "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-alt44", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-alt44", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "min_train_shard": 44, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-alt44", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "alt44" + ] + }, + "config_fingerprint": "60b9890a3df95e25", + "artifact_path": "experiments/think-d12-r20-alt44" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-alt44", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "60b9890a3df95e25" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 15, + "pos": 72944769, + "epoch": 1, + "pq_idx": 15, + "rg_idx": 72944769 + }, + "loop_state": { + "min_val_bpb": 1.1237689589285966, + "smooth_train_loss": 3.2428728970036347, + "total_training_time": 7990.569368600845, + "stage_training_flops": 1395284332511232000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1395284332511232000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_003500.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_003500.json new file mode 100644 index 0000000000000000000000000000000000000000..fda57c5444edc0f074c9bedb2a28edab2d912538 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/meta_003500.json @@ -0,0 +1,138 @@ +{ + "step": 3500, + "experiment_id": "think-d12-r20-alt44", + "val_bpb": 1.1019667576210812, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-alt44", + "wandb_run_id": "4b198ee6", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,alt44", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", + "experiment_id": "think-d12-r20-alt44", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", + "tokenizer_fingerprint": "3dc1a109161e9100", + "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-alt44", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-alt44", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "min_train_shard": 44, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-alt44", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "alt44" + ] + }, + "config_fingerprint": "60b9890a3df95e25", + "artifact_path": "experiments/think-d12-r20-alt44" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-alt44", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "60b9890a3df95e25" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 18, + "pos": 35096769, + "epoch": 1, + "pq_idx": 18, + "rg_idx": 35096769 + }, + "loop_state": { + "min_val_bpb": 1.1019667576210812, + "smooth_train_loss": 3.1139676059158647, + "total_training_time": 9326.703073263168, + "stage_training_flops": 1627831721263104000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1627831721263104000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_004000.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_004000.json new file mode 100644 index 0000000000000000000000000000000000000000..9f0591d0eda955c67b4b54b6419b7bc1358cce86 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/meta_004000.json @@ -0,0 +1,138 @@ +{ + "step": 4000, + "experiment_id": "think-d12-r20-alt44", + "val_bpb": 1.0801314281850647, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-alt44", + "wandb_run_id": "4b198ee6", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,alt44", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", + "experiment_id": "think-d12-r20-alt44", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", + "tokenizer_fingerprint": "3dc1a109161e9100", + "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-alt44", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-alt44", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "min_train_shard": 44, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-alt44", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "alt44" + ] + }, + "config_fingerprint": "60b9890a3df95e25", + "artifact_path": "experiments/think-d12-r20-alt44" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-alt44", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "60b9890a3df95e25" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 20, + "pos": 97248769, + "epoch": 1, + "pq_idx": 20, + "rg_idx": 97248769 + }, + "loop_state": { + "min_val_bpb": 1.0801314281850647, + "smooth_train_loss": 2.9255487986998965, + "total_training_time": 10663.225820541382, + "stage_training_flops": 1860379110014976000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1860379110014976000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/meta_004200.json b/experiments/think-d12-r20-alt44/base_checkpoints/meta_004200.json new file mode 100644 index 0000000000000000000000000000000000000000..e1b25dcb05e3f75b17da43f5e55ed0fa0d3294a2 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/meta_004200.json @@ -0,0 +1,138 @@ +{ + "step": 4200, + "experiment_id": "think-d12-r20-alt44", + "val_bpb": 1.0739757134682362, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-alt44", + "wandb_run_id": "4b198ee6", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio20,alt44", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20-alt44/base_checkpoints", + "experiment_id": "think-d12-r20-alt44", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20-alt44/config.json", + "tokenizer_fingerprint": "3dc1a109161e9100", + "git_commit_sha": "6e0a877d0c1bdd3f09b096b1a496e75cece29375", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-r20-alt44", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-alt44", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "min_train_shard": 44, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-alt44", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "alt44" + ] + }, + "config_fingerprint": "60b9890a3df95e25", + "artifact_path": "experiments/think-d12-r20-alt44" + }, + "stage": "base", + "base_experiment_id": "think-d12-r20-alt44", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "60b9890a3df95e25" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 22, + "pos": 2109569, + "epoch": 1, + "pq_idx": 22, + "rg_idx": 2109569 + }, + "loop_state": { + "min_val_bpb": 1.0739757134682362, + "smooth_train_loss": 2.961268984708144, + "total_training_time": 11197.636986970901, + "stage_training_flops": 1953398065515724800, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1953398065515724800 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_000500.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_000500.pt new file mode 100644 index 0000000000000000000000000000000000000000..d5466ef7f260b658cffd4b21f62676e679e4d8fb --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/model_000500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bc338f9658612bb2c61bcb0c3abf43b11fa1c77e1d4cd2a2cc6d2344d698eff7 +size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_001000.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_001000.pt new file mode 100644 index 0000000000000000000000000000000000000000..9cfe46772eb93c556f4bc704e081e86c2337951e --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/model_001000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daddc52f2e1f2bfc66a3667de7930a609b1cf5a961f3c43e70eb4ecbc1bf66f2 +size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_001500.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_001500.pt new file mode 100644 index 0000000000000000000000000000000000000000..95cc350408dac5c2906e9fdc8a053009995ce851 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/model_001500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:89de4208cd428992f1843c9573b2d265eb626aed058822e1bf67e31ace61e68a +size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_002000.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_002000.pt new file mode 100644 index 0000000000000000000000000000000000000000..8f5841b8531850042cfe1a7aa7d86ac166ebd2c5 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/model_002000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8e7dba25d5d192e349efe182eeb2d9e3869da546db073e0f08ec206c60af4cf1 +size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_002500.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_002500.pt new file mode 100644 index 0000000000000000000000000000000000000000..bd321fb95d5aa946677b94b5e1b182b21075c56d --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/model_002500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3c534b557b77646330a8ccea71f1ed5be71a26177f424604bd294389290ae13c +size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_003000.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_003000.pt new file mode 100644 index 0000000000000000000000000000000000000000..b4e33ee84a1952fb549830709f963842fb7a5a2c --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/model_003000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:20cdd57998ef483d11efc442342d56cbcd6f2b1bfa2f2d94d720495126f1ad26 +size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_003500.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_003500.pt new file mode 100644 index 0000000000000000000000000000000000000000..c574a2b464f72abdd6ea0ca90dfb42f7c14b3390 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/model_003500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ee39c24b9abf8cb1fd56ecd724c9c75a2234317549efd608cf18a5ea7e08f867 +size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_004000.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_004000.pt new file mode 100644 index 0000000000000000000000000000000000000000..3260d82ba2cd56a41f45ae4b89460346f2bdb77a --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/model_004000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9e2e3318b7c936109b78275bd929933c8b641b4d627cb8d61ceb6a15fb1d6696 +size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/model_004200.pt b/experiments/think-d12-r20-alt44/base_checkpoints/model_004200.pt new file mode 100644 index 0000000000000000000000000000000000000000..f4fcff7fe4833a12f96b964c65cf86cd83908d7b --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/model_004200.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b67a3866f260b405a9a17a766c684c22fcd5bacd68ac9fc5e4a2b9706c3672f0 +size 792761690 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_000500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..9d0cf69810eaf680e28f4d6f8accc07d16f2fe88 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/optim_000500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:83686f19e2a36223f557c5b809a0a4dc6b26ed96d9f3fec1998ad7feb6d83c81 +size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_001000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..97093521ce8bd53d1bbad8121aa98de585ca2a51 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/optim_001000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cfd90df1f6664d71a8b3804403404bd17bb67624aea05960fcbfc80d457dc225 +size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_001500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..d66e6a3f7aee939726dd46616f40a949835fad9c --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/optim_001500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4cfb2381d2c19cd5685c4b568ee805a54cf53503606533bd5d0fab11689cb661 +size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_002000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..143c7d3ce6f9b0ddf9a00922fb860a706c54e930 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/optim_002000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ca2c999553741ab8b6ab1e9243d633e42c02a77d5d919699180673f7362fc1d8 +size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_002500_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_002500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..6f74faaa86f3ddad97d1da35972a69e2b648d0ba --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/optim_002500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2ecc710a1ad4a3e4840c3b856bc1c97b524de991db5e143728ac1699946ea783 +size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_003000_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_003000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..b34099279b728ed8d7b40d68c75515ebd92709de --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/optim_003000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:93ccf45bb18ce1a8a28de29d2367789a62c4ea00dd178af88e88c6cf1c751bab +size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_003500_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_003500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..cc1ea143bf9e07227506a340e7b3d23012138b75 --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/optim_003500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bef09d5f4458358e4f0c5fc3c591d14f6fcc48cedcac6fb5d6ba28c8576024bc +size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_004000_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_004000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..864c9995e4fecbdf3b77b65fe379fb85d1e5ac1d --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/optim_004000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:273102ab4d2ef106bf25091f57305ebac76387952177e5fe9019b89c08f206bd +size 1246165357 diff --git a/experiments/think-d12-r20-alt44/base_checkpoints/optim_004200_rank0.pt b/experiments/think-d12-r20-alt44/base_checkpoints/optim_004200_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..0b789be8e232ccf18e6adf52baf62252a6520e6d --- /dev/null +++ b/experiments/think-d12-r20-alt44/base_checkpoints/optim_004200_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6ab30b78fc55f66d74ad106c4df7fbca65abf43b7cf1c8f52ba8b096500d1ce5 +size 1246165357 diff --git a/experiments/think-d12-r20-alt44/config.json b/experiments/think-d12-r20-alt44/config.json new file mode 100644 index 0000000000000000000000000000000000000000..a7094f72d48473b7398602d324ccae11d0490929 --- /dev/null +++ b/experiments/think-d12-r20-alt44/config.json @@ -0,0 +1,57 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20-alt44", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "min_train_shard": 44, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20-alt44", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20", + "alt44" + ] + }, + "config_fingerprint": "60b9890a3df95e25", + "artifact_path": "experiments/think-d12-r20-alt44" +} diff --git a/experiments/think-d12-r20-alt44/evals/core.json b/experiments/think-d12-r20-alt44/evals/core.json new file mode 100644 index 0000000000000000000000000000000000000000..eed98247e664236d9fd0b67615d50dae8d0788d3 --- /dev/null +++ b/experiments/think-d12-r20-alt44/evals/core.json @@ -0,0 +1,56 @@ +{ + "model": "base_model (step 4200)", + "step": 4200, + "bpb": {}, + "core_metric": 0.0774704140744213, + "core_results": { + "hellaswag_zeroshot": 0.2824138402938843, + "jeopardy": 0.0004723665479104966, + "bigbench_qa_wikidata": 0.057133015245199203, + "arc_easy": 0.32407405972480774, + "arc_challenge": 0.22098974883556366, + "copa": 0.5299999713897705, + "commonsense_qa": 0.3030303120613098, + "piqa": 0.5516865849494934, + "openbook_qa": 0.2460000067949295, + "lambada_openai": 0.22084222733974457, + "hellaswag": 0.2816171944141388, + "winograd": 0.5457875728607178, + "winogrande": 0.4956590235233307, + "bigbench_dyck_languages": 0.11100000888109207, + "agi_eval_lsat_ar": 0.2652173936367035, + "bigbench_cs_algorithms": 0.38181817531585693, + "bigbench_operators": 0.10952381044626236, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.03405865654349327, + "coqa": 0.0860578715801239, + "boolq": 0.593883752822876, + "bigbench_language_identification": 0.25049999356269836 + }, + "centered_results": { + "hellaswag_zeroshot": 0.04321845372517904, + "jeopardy": 0.0004723665479104966, + "bigbench_qa_wikidata": 0.057133015245199203, + "arc_easy": 0.09876541296641032, + "arc_challenge": -0.03868033488591512, + "copa": 0.059999942779541016, + "commonsense_qa": 0.12878789007663724, + "piqa": 0.10337316989898682, + "openbook_qa": -0.005333324273427327, + "lambada_openai": 0.22084222733974457, + "hellaswag": 0.04215625921885172, + "winograd": 0.09157514572143555, + "winogrande": -0.008681952953338623, + "bigbench_dyck_languages": 0.11100000888109207, + "agi_eval_lsat_ar": 0.08152174204587935, + "bigbench_cs_algorithms": 0.38181817531585693, + "bigbench_operators": 0.10952381044626236, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.03405865654349327, + "coqa": 0.0860578715801239, + "boolq": -0.06872696625558952, + "bigbench_language_identification": 0.1754675396729355 + }, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/evals/samples.json b/experiments/think-d12-r20-alt44/evals/samples.json new file mode 100644 index 0000000000000000000000000000000000000000..84b89fcbe14b41eea76fb2698f0cde99a3147e5c --- /dev/null +++ b/experiments/think-d12-r20-alt44/evals/samples.json @@ -0,0 +1,48 @@ +{ + "model": "base_model (step 4200)", + "step": 4200, + "bpb": {}, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is 10,000,000 francs, or 10," + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is the same as the symbol of the sun, and the same as the symbol of" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nIf yesterday was Saturday, then tomorrow will be Sunday. \n\n" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is a very common error, and is not a little remarkable. It is a very" + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, 2. The moon, 3. The" + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is a reddish brown, and the color of the skin is a redd" + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the square of the number of the square of the number of the" + } + ], + "unconditioned_samples": [ + "<|bos|>asi that of her reputed father. There were in her last year or two here so many words of reproach, accusing all her priests, waiting on her travail of life, that they called her true wife therefore, and implored for forgiveness. There was one again who, after so many curses, murmured induced hope through advantages, confidence derived by human hopes, thus she bethought her of the tempted profligate, bethinking her what she was doing; and there was a third person who, in attempting to induce her to accept either the fall or the punishment of the false accuser, vindicated herself from the foul", + "<|bos|> Carrapal, the lips visible, and, unlike \n\nLeblanc, its lips sinuous, and\n\n Rod consumed them and, what is more, they are ashes of poisons, and, to shut out therefrom the \n\n6 18k of a retail price, every part of his body has an appletree within his bowels, which they are said to destroy; his heart acts fiery compositions. \n\nRABINS. A narrow process on a chemical tree. \n\n1\u00ba 1008. GOLD AND SILVER BENCH. \n\n1179. BLACK and red ink, one in dye of black or black t", + "<|bos|>ANA COMMISSIONER'S EXPRESS-POWER OF STUDIES \n\nWants not to be shown in museum rulings without such warrant, as where a survey is not admissible as error to show the boundary of a\n\nKANSAS HISTORY \n\nKENTUCKY OPINIONS OR DOCUMENTS 23 \n\nREPORTS \n\nCOMPL \n\nJOHNSTON \n\nJOHNSON \n\nMORE \n\nJOHN GREEN \n\nST. LOUIS \n\nJOHNSTON \n\nTRAPP \n\nLOVE \n\nLOUISE LOAKE \n\n ANDORE \n\nChastity de- with was reported by the Superintendent of Education opinion that Washington was the nearest approximation Vincennes, the wisdom merchandise to do the actual route to to coalfields is", + "<|bos|> wield hard 160,6 G.C.M \n\n3.1\n\n10439 An Introduction \n\nFind a good Equinoxa (S.P.M.C.C.) \n\nFind a good\n\nEquinoxa \n\nNa (S.P.M.C.C.) \n\nAnt oc \"Soft may the light serene \n\nCome through to us that liveth \n\nSweet air of heaven, and while flutes \n\nBurn on the Sabbaths that thy soul \n\nBlows with unspeakable bliss\" (S.P.C.) Fra\n\n\u0399\u0399 Chant chosen rather than a plowbox) \n\n11 be the good gentle wind", + "<|bos|> Lac they are still four small animals which deserve to be treated as tame things.' \n\nFor making them wild I have also made in three mounds the beating of large pipes, a process in which Paul Ursula is interribted with a small fsel, or rather escape of the mauds: will,' says Augustine, 'only \n\n1 Hist. Ancient Cities, iii, 4.\n\nWAORIES IN BAB LANGUAGES \n\nmuch trouble.' For the range and torridity of the Mediterranean they which chiefly merit attention are these: (I.) the solitary growth of the baize trees thrown out by the Tartars from Aujib", + "<|bos|>PLmillan, 1,433), and p. 758. Ay, now tell me, but let him tell thee v.: III.\n\nI must apply it to touch the where he has desired to see it. It need hardly be said, that in the nursery, Diogenes seeing the anachronymus hesitates: or, in the pining out of Franklin's poisoned spider, passage \u00e4n his retort to the school-house, was saying: \"I cannot shave better than I saved from drowning, and knocked out again from living ten years, and rained down curds", + "<|bos|>1749.11 \n\n17434721S+ Vpels during 1749), and again at the end of the year 7. 9 1\n\n2044 020 035 844 \n\nStowe, 91", + "<|bos|> PEOPLE MEET OTHERS IN THEIT\u00c9 OF A GREAT BUNTER HILL \n\nTHE \n\nTHE READERS THAT FORTUNATELY MEET\n\nTWO OCCURRENCES NEAR THE TROAD NICOLO JAM THE BRITISH \n\nThe well-known type of New England ferry-boat presented to Captain John Chapman Wilson some years ago has been very generally appreciated (or ignored) and a current of eighty yards was paid down by Professor\n\nCleghorn as 54 cents. This was his generous way of paying anything to a schoolmate under any circumstances. When Capt. John Chapman first tried to reach the shore of St. Laurent in Camp Bellingham, he did not" + ] +} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/evals/val_bpb.json b/experiments/think-d12-r20-alt44/evals/val_bpb.json new file mode 100644 index 0000000000000000000000000000000000000000..ebeb5f9a060a85d65eaddf873258e61696678699 --- /dev/null +++ b/experiments/think-d12-r20-alt44/evals/val_bpb.json @@ -0,0 +1,12 @@ +{ + "model": "base_model (step 4200)", + "step": 4200, + "bpb": { + "val": 1.0180559919724057 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/think-d12-r20-alt44/run.json b/experiments/think-d12-r20-alt44/run.json new file mode 100644 index 0000000000000000000000000000000000000000..33d70e05fbc50d65a397a92f05c178475878ef88 --- /dev/null +++ b/experiments/think-d12-r20-alt44/run.json @@ -0,0 +1,10 @@ +{ + "experiment_id": "think-d12-r20-alt44", + "stage": "base", + "base_experiment_id": "think-d12-r20-alt44", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "60b9890a3df95e25", + "wandb_run_id": "4b198ee6", + "created_at": 1781721383 +} diff --git a/experiments/think-d12-r20-alt44/summary.json b/experiments/think-d12-r20-alt44/summary.json new file mode 100644 index 0000000000000000000000000000000000000000..b6c93c2f8d686d561ff5747d6010390ef0e047f8 --- /dev/null +++ b/experiments/think-d12-r20-alt44/summary.json @@ -0,0 +1,93 @@ +{ + "experiment_id": "think-d12-r20-alt44", + "stage": "base", + "base_experiment_id": "think-d12-r20-alt44", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "dataset": "jbduran/think-dataset", + "dataset_revision": "main", + "step": 4200, + "depth": 12, + "target_param_data_ratio": 20.0, + "training_tokens": 2202009600, + "final_sampled_val_bpb": 1.0739757134682362, + "minimum_sampled_val_bpb": 1.0739757134682362, + "full_val_bpb": 1.0180559919724057, + "core_metric": 0.0774704140744213, + "centered_results": { + "hellaswag_zeroshot": 0.04321845372517904, + "jeopardy": 0.0004723665479104966, + "bigbench_qa_wikidata": 0.057133015245199203, + "arc_easy": 0.09876541296641032, + "arc_challenge": -0.03868033488591512, + "copa": 0.059999942779541016, + "commonsense_qa": 0.12878789007663724, + "piqa": 0.10337316989898682, + "openbook_qa": -0.005333324273427327, + "lambada_openai": 0.22084222733974457, + "hellaswag": 0.04215625921885172, + "winograd": 0.09157514572143555, + "winogrande": -0.008681952953338623, + "bigbench_dyck_languages": 0.11100000888109207, + "agi_eval_lsat_ar": 0.08152174204587935, + "bigbench_cs_algorithms": 0.38181817531585693, + "bigbench_operators": 0.10952381044626236, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.03405865654349327, + "coqa": 0.0860578715801239, + "boolq": -0.06872696625558952, + "bigbench_language_identification": 0.1754675396729355 + }, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is 10,000,000 francs, or 10," + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is the same as the symbol of the sun, and the same as the symbol of" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. \n\nIf yesterday was Saturday, then tomorrow will be Sunday. \n\n" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is a very common error, and is not a little remarkable. It is a very" + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: \n\n1. The sun, 2. The moon, 3. The" + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is a reddish brown, and the color of the skin is a redd" + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is the number of the square of the number of the square of the number of the" + } + ], + "unconditioned_samples": [ + "<|bos|>asi that of her reputed father. There were in her last year or two here so many words of reproach, accusing all her priests, waiting on her travail of life, that they called her true wife therefore, and implored for forgiveness. There was one again who, after so many curses, murmured induced hope through advantages, confidence derived by human hopes, thus she bethought her of the tempted profligate, bethinking her what she was doing; and there was a third person who, in attempting to induce her to accept either the fall or the punishment of the false accuser, vindicated herself from the foul", + "<|bos|> Carrapal, the lips visible, and, unlike \n\nLeblanc, its lips sinuous, and\n\n Rod consumed them and, what is more, they are ashes of poisons, and, to shut out therefrom the \n\n6 18k of a retail price, every part of his body has an appletree within his bowels, which they are said to destroy; his heart acts fiery compositions. \n\nRABINS. A narrow process on a chemical tree. \n\n1\u00ba 1008. GOLD AND SILVER BENCH. \n\n1179. BLACK and red ink, one in dye of black or black t", + "<|bos|>ANA COMMISSIONER'S EXPRESS-POWER OF STUDIES \n\nWants not to be shown in museum rulings without such warrant, as where a survey is not admissible as error to show the boundary of a\n\nKANSAS HISTORY \n\nKENTUCKY OPINIONS OR DOCUMENTS 23 \n\nREPORTS \n\nCOMPL \n\nJOHNSTON \n\nJOHNSON \n\nMORE \n\nJOHN GREEN \n\nST. LOUIS \n\nJOHNSTON \n\nTRAPP \n\nLOVE \n\nLOUISE LOAKE \n\n ANDORE \n\nChastity de- with was reported by the Superintendent of Education opinion that Washington was the nearest approximation Vincennes, the wisdom merchandise to do the actual route to to coalfields is", + "<|bos|> wield hard 160,6 G.C.M \n\n3.1\n\n10439 An Introduction \n\nFind a good Equinoxa (S.P.M.C.C.) \n\nFind a good\n\nEquinoxa \n\nNa (S.P.M.C.C.) \n\nAnt oc \"Soft may the light serene \n\nCome through to us that liveth \n\nSweet air of heaven, and while flutes \n\nBurn on the Sabbaths that thy soul \n\nBlows with unspeakable bliss\" (S.P.C.) Fra\n\n\u0399\u0399 Chant chosen rather than a plowbox) \n\n11 be the good gentle wind", + "<|bos|> Lac they are still four small animals which deserve to be treated as tame things.' \n\nFor making them wild I have also made in three mounds the beating of large pipes, a process in which Paul Ursula is interribted with a small fsel, or rather escape of the mauds: will,' says Augustine, 'only \n\n1 Hist. Ancient Cities, iii, 4.\n\nWAORIES IN BAB LANGUAGES \n\nmuch trouble.' For the range and torridity of the Mediterranean they which chiefly merit attention are these: (I.) the solitary growth of the baize trees thrown out by the Tartars from Aujib", + "<|bos|>PLmillan, 1,433), and p. 758. Ay, now tell me, but let him tell thee v.: III.\n\nI must apply it to touch the where he has desired to see it. It need hardly be said, that in the nursery, Diogenes seeing the anachronymus hesitates: or, in the pining out of Franklin's poisoned spider, passage \u00e4n his retort to the school-house, was saying: \"I cannot shave better than I saved from drowning, and knocked out again from living ten years, and rained down curds", + "<|bos|>1749.11 \n\n17434721S+ Vpels during 1749), and again at the end of the year 7. 9 1\n\n2044 020 035 844 \n\nStowe, 91", + "<|bos|> PEOPLE MEET OTHERS IN THEIT\u00c9 OF A GREAT BUNTER HILL \n\nTHE \n\nTHE READERS THAT FORTUNATELY MEET\n\nTWO OCCURRENCES NEAR THE TROAD NICOLO JAM THE BRITISH \n\nThe well-known type of New England ferry-boat presented to Captain John Chapman Wilson some years ago has been very generally appreciated (or ignored) and a current of eighty yards was paid down by Professor\n\nCleghorn as 54 cents. This was his generous way of paying anything to a schoolmate under any circumstances. When Capt. John Chapman first tried to reach the shore of St. Laurent in Camp Bellingham, he did not" + ], + "training_time_seconds": 11197.636986970901, + "stage_training_flops": 1.9533980655157248e+18, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1.9533980655157248e+18, + "config_fingerprint": "60b9890a3df95e25", + "git_commit_sha": "0ea306100c54d5e0950d28c092ec5291ee71c247", + "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/4b198ee6", + "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r20-alt44", + "dataset_fingerprint": "147853d67cb3de71", + "tokenizer_fingerprint": "3dc1a109161e9100", + "unique_train_tokens": 2268069888, + "effective_epochs": 0.970873786407767 +} diff --git a/experiments/think-d12-r20-alt44/tokenizer/experiment_tokenizer.json b/experiments/think-d12-r20-alt44/tokenizer/experiment_tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..1f277b2e7847d48764700614f968e2ebad2bd276 --- /dev/null +++ b/experiments/think-d12-r20-alt44/tokenizer/experiment_tokenizer.json @@ -0,0 +1,19 @@ +{ + "experiment_id": "think-d12-r20-alt44", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "min_train_shard": 44, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "created_at": 1781721409 +} diff --git a/experiments/think-d12-r20-alt44/tokenizer/token_bytes.pt b/experiments/think-d12-r20-alt44/tokenizer/token_bytes.pt new file mode 100644 index 0000000000000000000000000000000000000000..d5505db6f555c8497724954a1dd310ecfcf52b3b --- /dev/null +++ b/experiments/think-d12-r20-alt44/tokenizer/token_bytes.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:271a4064bad6183970392e829ff7a8fd25ba37a811b880fde5f5b6d58823a5a0 +size 132649 diff --git a/experiments/think-d12-r20-alt44/tokenizer/tokenizer.pkl b/experiments/think-d12-r20-alt44/tokenizer/tokenizer.pkl new file mode 100644 index 0000000000000000000000000000000000000000..bb8aaabb88ba0c6eb3d6efe0e26704772111316e --- /dev/null +++ b/experiments/think-d12-r20-alt44/tokenizer/tokenizer.pkl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b99e5e10fbe663212046c441fd4d584a19c1a5531e8ebea04f8e42238998a4e4 +size 403342 diff --git a/experiments/think-d12-r20/base_checkpoints/meta_000500.json b/experiments/think-d12-r20/base_checkpoints/meta_000500.json new file mode 100644 index 0000000000000000000000000000000000000000..ba79b4dc95de94cd09f87814f615475ece474c1e --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/meta_000500.json @@ -0,0 +1,61 @@ +{ + "step": 500, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "d12-ratio20" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 2, + "pos": 62184769, + "epoch": 1, + "pq_idx": 2, + "rg_idx": 62184769 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 3.5891939434241213, + "total_training_time": 1312.64857006073 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_001000.json b/experiments/think-d12-r20/base_checkpoints/meta_001000.json new file mode 100644 index 0000000000000000000000000000000000000000..bbca0e8342d8f36e4a06b1933ce176471b508a07 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/meta_001000.json @@ -0,0 +1,61 @@ +{ + "step": 1000, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "d12-ratio20" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 5, + "pos": 24336769, + "epoch": 1, + "pq_idx": 5, + "rg_idx": 24336769 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 3.4868806829553485, + "total_training_time": 2654.1426842212677 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_001500.json b/experiments/think-d12-r20/base_checkpoints/meta_001500.json new file mode 100644 index 0000000000000000000000000000000000000000..4d052d841af639e6ed386c3bf952cdef76107cbf --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/meta_001500.json @@ -0,0 +1,61 @@ +{ + "step": 1500, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "d12-ratio20" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 7, + "pos": 86488769, + "epoch": 1, + "pq_idx": 7, + "rg_idx": 86488769 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 3.410999880107203, + "total_training_time": 3995.465323448181 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_002000.json b/experiments/think-d12-r20/base_checkpoints/meta_002000.json new file mode 100644 index 0000000000000000000000000000000000000000..8ccecb680019f8c6b8dd9c312d17235d307c95bf --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/meta_002000.json @@ -0,0 +1,61 @@ +{ + "step": 2000, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "d12-ratio20" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 10, + "pos": 48640769, + "epoch": 1, + "pq_idx": 10, + "rg_idx": 48640769 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 3.452807694528099, + "total_training_time": 5336.5537366867065 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_002500.json b/experiments/think-d12-r20/base_checkpoints/meta_002500.json new file mode 100644 index 0000000000000000000000000000000000000000..96fc7e712e052057bcc504c98aa1891ccee8b9e2 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/meta_002500.json @@ -0,0 +1,61 @@ +{ + "step": 2500, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "d12-ratio20" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 13, + "pos": 10792769, + "epoch": 1, + "pq_idx": 13, + "rg_idx": 10792769 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 3.327396609246394, + "total_training_time": 6677.6987290382385 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_003000.json b/experiments/think-d12-r20/base_checkpoints/meta_003000.json new file mode 100644 index 0000000000000000000000000000000000000000..7ea0f8f11e2301baef3d24fa3c60d0fa0a52f9de --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/meta_003000.json @@ -0,0 +1,61 @@ +{ + "step": 3000, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "d12-ratio20" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 15, + "pos": 72944769, + "epoch": 1, + "pq_idx": 15, + "rg_idx": 72944769 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 3.048554485031434, + "total_training_time": 8018.697687149048 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_003500.json b/experiments/think-d12-r20/base_checkpoints/meta_003500.json new file mode 100644 index 0000000000000000000000000000000000000000..ea7c74460fdec7160133b9cbc32e9e20470153ee --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/meta_003500.json @@ -0,0 +1,61 @@ +{ + "step": 3500, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "d12-ratio20" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 18, + "pos": 35096769, + "epoch": 1, + "pq_idx": 18, + "rg_idx": 35096769 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 3.0863692227626416, + "total_training_time": 9359.67183303833 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_004000.json b/experiments/think-d12-r20/base_checkpoints/meta_004000.json new file mode 100644 index 0000000000000000000000000000000000000000..e43944a04fee67ab2d03020bbd670e2b024bf741 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/meta_004000.json @@ -0,0 +1,61 @@ +{ + "step": 4000, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "d12-ratio20" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 20, + "pos": 97248769, + "epoch": 1, + "pq_idx": 20, + "rg_idx": 97248769 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 2.7993042062101186, + "total_training_time": 10700.389838218689 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/meta_004200.json b/experiments/think-d12-r20/base_checkpoints/meta_004200.json new file mode 100644 index 0000000000000000000000000000000000000000..edcf82bc7f5d7ab29272189b9203fe91c518bbe2 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/meta_004200.json @@ -0,0 +1,61 @@ +{ + "step": 4200, + "val_bpb": null, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "dummy", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": -1, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "eval_every": -1, + "eval_tokens": 41943040, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "d12-ratio20" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 22, + "pos": 2109569, + "epoch": 1, + "pq_idx": 22, + "rg_idx": 2109569 + }, + "loop_state": { + "min_val_bpb": Infinity, + "smooth_train_loss": 2.8461822660242255, + "total_training_time": 11236.57096004486 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20/base_checkpoints/model_000500.pt b/experiments/think-d12-r20/base_checkpoints/model_000500.pt new file mode 100644 index 0000000000000000000000000000000000000000..97e4c71cc3328cef1fabf54e4a1b8ecf3e240e32 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/model_000500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:00e14e7ad4637b08c6e5f649ebc7de7ed0d9fcb39a293b7b088a13c2ebb549f8 +size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_001000.pt b/experiments/think-d12-r20/base_checkpoints/model_001000.pt new file mode 100644 index 0000000000000000000000000000000000000000..e2d60950935b1958573cb2de2aa7a12151bc1f16 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/model_001000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:01e2a9466841de94c8c3a654a257c842c1764d3ac4cd79cf478f9f2b24859e7c +size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_001500.pt b/experiments/think-d12-r20/base_checkpoints/model_001500.pt new file mode 100644 index 0000000000000000000000000000000000000000..5bad12b7a4a0a3593d00a4ba4d742ef0bd7e5945 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/model_001500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:03fb5d4241142f86b69396fe200295929df1b4bcc2ba5d4cc464c9627cf0a530 +size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_002000.pt b/experiments/think-d12-r20/base_checkpoints/model_002000.pt new file mode 100644 index 0000000000000000000000000000000000000000..283db3b4a5e606111e288f51ab830fceb5ea3fb3 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/model_002000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c4db8d9fa2b5fa4ca14f65dfd8d5356c691c5c8772e839555f0fec6f8d300269 +size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_002500.pt b/experiments/think-d12-r20/base_checkpoints/model_002500.pt new file mode 100644 index 0000000000000000000000000000000000000000..766aefe9911a42ba0478ae0effee731d7e248fe1 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/model_002500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:abb54590a71621be44e418b133a42088aed1d84872a9726e9aa4ac8b7e418788 +size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_003000.pt b/experiments/think-d12-r20/base_checkpoints/model_003000.pt new file mode 100644 index 0000000000000000000000000000000000000000..95408589ebb4a00a04f89c866519d954a17a96db --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/model_003000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0a96aa9105d455f741602d43d807dd0b2732823c9c8be3465717f7090aa469eb +size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_003500.pt b/experiments/think-d12-r20/base_checkpoints/model_003500.pt new file mode 100644 index 0000000000000000000000000000000000000000..aa7696886c6993d18f98632b2a12d2e93c89c890 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/model_003500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bab4a0741306ce5945c9a86dabd233081b836f0e1f814dd1d9363cc1b9cce950 +size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_004000.pt b/experiments/think-d12-r20/base_checkpoints/model_004000.pt new file mode 100644 index 0000000000000000000000000000000000000000..195c00c1e8c9542b4c3a4237f783261f127da27b --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/model_004000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6c967e927215ebca6af3f49011d98fd8fa4487c55e78485271878e382b7293d5 +size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/model_004200.pt b/experiments/think-d12-r20/base_checkpoints/model_004200.pt new file mode 100644 index 0000000000000000000000000000000000000000..c201085206abc74184928a0dac513e531432f6cc --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/model_004200.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1dfab1c19f9bbd3dd27ed859b70cfd242d5ebaef734985dbef78d4998493a6f6 +size 792761399 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_000500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..3c6f2f796a2895cf54026c01b22c14a52177d24e --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/optim_000500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:62b942e737b779c2dbb3b14246ffc951fb9dce816e10e7ada936b81093faaca7 +size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_001000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..0272991a5a4536ff457d456339de4d750c8cda24 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/optim_001000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f5fe23783669dcb79dda4bc0a66e2c536d9d3055fbcaf07840662e2c5b8f6950 +size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_001500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..88a0adfe1b99ce52dcf0807bf3881371f4ed1184 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/optim_001500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9dcce76cabc6cdd8083af5f064ae391d8b84fc96d2608ab6740fffc2b1e2f5f2 +size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_002000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..b11fbe743399c95fabbd0696c2852a595104d922 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/optim_002000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4dc97c55d9aa7e22cd38dcfc298b42066aeda741daf7d1b508575ac62c0e32de +size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_002500_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_002500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..37d285b746991053e40d3e97b32448203c24c826 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/optim_002500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2550badb9d208d9f2e51765bafd6789cc2f9ee8b6fd5880e25790d427deb9556 +size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_003000_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_003000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..b78dee7e866f9e7c539a9d6ca1e48264d7b2c2b5 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/optim_003000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8ea5ec19b0375980f64cbdbb9709d040a01e2f546ebdcc380246f8c69d26b242 +size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_003500_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_003500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..26f63d83d8e08a5bf8c6b3139f798f5dabfb10aa --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/optim_003500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9465372371e5d248972fe855b83234cbf5fde5bcf208eb7e2faf9f51cf207def +size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_004000_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_004000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..8ad1e5577fd1f8636688a15b9f932609943770e6 --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/optim_004000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c27aa4c1fea51a61cd5aef3654780852040f59b347b9f18c7a9f29620664b677 +size 1246165237 diff --git a/experiments/think-d12-r20/base_checkpoints/optim_004200_rank0.pt b/experiments/think-d12-r20/base_checkpoints/optim_004200_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..dcfacec3437fca29f63311b48f8f86a813607bcd --- /dev/null +++ b/experiments/think-d12-r20/base_checkpoints/optim_004200_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3a71d1f755bf5e13a1bd7741a3a1bbc1d9807468b5190c4e2380b91e0116eee0 +size 1246165237 diff --git a/experiments/think-d12-r20/config.json b/experiments/think-d12-r20/config.json new file mode 100644 index 0000000000000000000000000000000000000000..ca4f4971460638cb8ce3e0797217d0dcd2781f49 --- /dev/null +++ b/experiments/think-d12-r20/config.json @@ -0,0 +1,55 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-r20", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio20" + ] + }, + "config_fingerprint": "4236ec51a84df577", + "artifact_path": "experiments/think-d12-r20" +} diff --git a/experiments/think-d12-r20/evals/val_bpb.json b/experiments/think-d12-r20/evals/val_bpb.json new file mode 100644 index 0000000000000000000000000000000000000000..a3685444ee7259a3f6389b3947e141787e52031b --- /dev/null +++ b/experiments/think-d12-r20/evals/val_bpb.json @@ -0,0 +1,12 @@ +{ + "model": "base_model (step 4200)", + "step": 4200, + "bpb": { + "val": 1.0597927744819549 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/think-d12-r20/run.json b/experiments/think-d12-r20/run.json new file mode 100644 index 0000000000000000000000000000000000000000..c1a7fefa17921b8f72b628a7bcc3d2ceb1842236 --- /dev/null +++ b/experiments/think-d12-r20/run.json @@ -0,0 +1,6 @@ +{ + "experiment_id": "think-d12-r20", + "stage": "base", + "wandb_run_id": null, + "migration_note": "Migrated from the pre-lineage repository layout." +} diff --git a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/meta_000015.json b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/meta_000015.json new file mode 100644 index 0000000000000000000000000000000000000000..0ed9a3f691091d60e600b73aefd16fda1c178857 --- /dev/null +++ b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/meta_000015.json @@ -0,0 +1,102 @@ +{ + "step": 15, + "val_bpb": 0.9148607229178501, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-r20-pre1930-authentic", + "wandb_run_id": "f1457d9c", + "wandb_group": "think-d12", + "wandb_tags": "sft,pre1930,ratio20", + "device_type": "", + "model_tag": null, + "model_step": null, + "base_checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20/base_checkpoints", + "base_step": 4200, + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-r20/tokenizer", + "resume_from_step": null, + "experiment_id": "think-d12-r20-pre1930-authentic", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/config.json", + "parent_cumulative_flops": 1.95339809193984e+18, + "tokenizer_fingerprint": "6e592b9b323f98bf", + "git_commit_sha": "16a49c72218157b03f4bb238a0fdc23f0bca8b18", + "load_optimizer": 1, + "num_iterations": -1, + "max_seq_len": null, + "device_batch_size": 8, + "total_batch_size": null, + "embedding_lr": null, + "unembedding_lr": null, + "matrix_lr": null, + "init_lr_frac": 0.8, + "warmup_ratio": 0.0, + "warmdown_ratio": 0.5, + "final_lr_frac": 0.0, + "eval_every": -1, + "eval_tokens": 20971520, + "chatcore_every": -1, + "chatcore_max_cat": -1, + "chatcore_max_sample": 24, + "save_every": -1, + "recipe": "pre1930", + "pre1930_epochs": 5, + "mmlu_epochs": 3, + "gsm8k_epochs": 4, + "resolved_experiment_config": { + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-authentic", + "data": { + "recipe": "pre1930", + "pre1930_epochs": 5 + }, + "training": { + "num_iterations": -1, + "device_batch_size": 8, + "eval_every": -1, + "chatcore_every": -1, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "pre1930", + "ratio20" + ] + }, + "config_fingerprint": "d3378357cef17359", + "artifact_path": "experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic" + }, + "stage": "sft", + "base_experiment_id": null, + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "d3378357cef17359" + }, + "loop_state": { + "step": 15, + "total_training_time": 14.16254210472107, + "min_val_bpb": 0.9148607229178501, + "smooth_train_loss": 1.5855611060774228, + "mfu": 52.60094494392411, + "tok_per_sec": 185002, + "stage_training_flops": 6976421662556160.0, + "inherited_parent_flops": 1.95339809193984e+18, + "cumulative_pipeline_training_flops": 1.9603745136023962e+18 + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/model_000015.pt b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/model_000015.pt new file mode 100644 index 0000000000000000000000000000000000000000..dde9bb22e5578ccf650a6ac7c901146a455916af --- /dev/null +++ b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/model_000015.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4893fa6fd812a1b5ec282a174f3c7bd6df69ce5f73d785e59537e2378bc2c833 +size 792761690 diff --git a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/optim_000015_rank0.pt b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/optim_000015_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..a91d44ccdb7da88396f21313673e1405d10feadb --- /dev/null +++ b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/checkpoints/optim_000015_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4d72645c73b58777d4cef8e50f304b632f4dd09f7bb023f985ba458c19669207 +size 1246165357 diff --git a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/config.json b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/config.json new file mode 100644 index 0000000000000000000000000000000000000000..a8d72cacc5c7fa2896dc5423493c9298441b22c1 --- /dev/null +++ b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/config.json @@ -0,0 +1,32 @@ +{ + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-authentic", + "data": { + "recipe": "pre1930", + "pre1930_epochs": 5 + }, + "training": { + "num_iterations": -1, + "device_batch_size": 8, + "eval_every": -1, + "chatcore_every": -1, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "pre1930", + "ratio20" + ] + }, + "config_fingerprint": "d3378357cef17359", + "artifact_path": "experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic" +} diff --git a/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/run.json b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/run.json new file mode 100644 index 0000000000000000000000000000000000000000..8716944f8a2466c00be953dc4c6746b800d8b0b9 --- /dev/null +++ b/experiments/think-d12-r20/sft/think-d12-r20-pre1930-authentic/run.json @@ -0,0 +1,10 @@ +{ + "experiment_id": "think-d12-r20-pre1930-authentic", + "stage": "sft", + "base_experiment_id": "think-d12-r20", + "parent_experiment_id": "think-d12-r20", + "parent_checkpoint_step": null, + "config_fingerprint": "d3378357cef17359", + "wandb_run_id": "f1457d9c", + "created_at": 1782154820 +} diff --git a/experiments/think-d12-r20/summary.json b/experiments/think-d12-r20/summary.json new file mode 100644 index 0000000000000000000000000000000000000000..6e881a9514eb34f23dea96541f8e7559ef9441a4 --- /dev/null +++ b/experiments/think-d12-r20/summary.json @@ -0,0 +1,30 @@ +{ + "experiment_id": "think-d12-r20", + "stage": "base", + "base_experiment_id": "think-d12-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "dataset": "jbduran/think-dataset", + "dataset_revision": "main", + "step": 4200, + "depth": 12, + "target_param_data_ratio": 20.0, + "training_tokens": 2202009600, + "final_sampled_val_bpb": null, + "minimum_sampled_val_bpb": Infinity, + "full_val_bpb": 1.0597927744819549, + "core_metric": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [], + "training_time_seconds": 11236.57096004486, + "stage_training_flops": 1.95339809193984e+18, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1.95339809193984e+18, + "config_fingerprint": "4236ec51a84df577", + "git_commit_sha": "44d6e64aec8e6a1ec2f2a621ca1b44574b705d66", + "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/None", + "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/think-d12-r20", + "dataset_fingerprint": "a6e1b3a100e0d8b3", + "tokenizer_fingerprint": "6e592b9b323f98bf" +} diff --git a/experiments/think-d12-r20/tokenizer/think_dataset_tokenizer.json b/experiments/think-d12-r20/tokenizer/think_dataset_tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..fea3141f8940ad53370916cd7a3dc8743edbf3ad --- /dev/null +++ b/experiments/think-d12-r20/tokenizer/think_dataset_tokenizer.json @@ -0,0 +1,20 @@ +{ + "dataset_repo": "jbduran/think-dataset", + "manifest_source_dataset": "institutional/institutional-books-1.0", + "num_train_shards": 24, + "expected_val_shard": "shard_00472.parquet", + "filters": { + "language": "eng", + "min_english_proportion": 0.9, + "year_max_exclusive": 1930, + "reject_invalid_date_types": true, + "ocr_min_inclusive": 90.0, + "ocr_disagreement_max_inclusive": 10.0, + "min_tokenizability": 95.0, + "min_tokens": 500, + "min_chars": 2000, + "min_pages": 3, + "min_sentences": 20, + "undated_rows_rejected": true + } +} \ No newline at end of file diff --git a/experiments/think-d12-r20/tokenizer/token_bytes.pt b/experiments/think-d12-r20/tokenizer/token_bytes.pt new file mode 100644 index 0000000000000000000000000000000000000000..01d1ec4aab9e8a7d205c3b3ffbeb8da0e9a62db1 --- /dev/null +++ b/experiments/think-d12-r20/tokenizer/token_bytes.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:53db25a1f6969198e67c303e479e97fb49859c715061baec525471666a2f47e1 +size 132649 diff --git a/experiments/think-d12-r20/tokenizer/tokenizer.pkl b/experiments/think-d12-r20/tokenizer/tokenizer.pkl new file mode 100644 index 0000000000000000000000000000000000000000..a17bd392980021628053b95d6425fc556aad527a --- /dev/null +++ b/experiments/think-d12-r20/tokenizer/tokenizer.pkl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:42ea52179dcefb47e4a4bbca7c37f67c3373840a3681587b1c8fe188f00d3bf1 +size 404071 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_000500.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_000500.json new file mode 100644 index 0000000000000000000000000000000000000000..76b983909f450a9f8bb62c3217333e4767152458 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_000500.json @@ -0,0 +1,139 @@ +{ + "step": 500, + "training_complete": false, + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "val_bpb": 1.332522614651975, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "thinkcleaned-d12-1ep-sh26-r11", + "wandb_run_id": "e8375954", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset-cleaned,d12,ratio11.25", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints", + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json", + "tokenizer_fingerprint": "0a9922b59b5cb78a", + "git_commit_sha": "7f4c957646ab9d9314d7a7aae4f7efd73aa1ff2e", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "thinkcleaned-d12-1ep-sh26-r11", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset-clean", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 26, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "seed": 42, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "thinkcleaned-d12-1ep-sh26-r11", + "group": "think-d12", + "tags": [ + "think-dataset-cleaned", + "d12", + "ratio11.25" + ] + }, + "config_fingerprint": "3d2656cfe49b6048", + "artifact_path": "experiments/thinkcleaned-d12-1ep-sh26-r11" + }, + "stage": "base", + "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "3d2656cfe49b6048" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 2, + "pos": 62184769, + "epoch": 1, + "pq_idx": 2, + "rg_idx": 62184769 + }, + "loop_state": { + "min_val_bpb": 1.332522614651975, + "smooth_train_loss": 3.7214987179114285, + "total_training_time": 1307.1219980716705, + "stage_training_flops": 232547388751872000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 232547388751872000 + } +} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_001000.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_001000.json new file mode 100644 index 0000000000000000000000000000000000000000..4f044ff47be8f5afc8d65a5eea38203559569714 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_001000.json @@ -0,0 +1,139 @@ +{ + "step": 1000, + "training_complete": false, + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "val_bpb": 1.2404084951472125, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "thinkcleaned-d12-1ep-sh26-r11", + "wandb_run_id": "e8375954", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset-cleaned,d12,ratio11.25", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": 500, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints", + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json", + "tokenizer_fingerprint": "0a9922b59b5cb78a", + "git_commit_sha": "7f4c957646ab9d9314d7a7aae4f7efd73aa1ff2e", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "thinkcleaned-d12-1ep-sh26-r11", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset-clean", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 26, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "seed": 42, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "thinkcleaned-d12-1ep-sh26-r11", + "group": "think-d12", + "tags": [ + "think-dataset-cleaned", + "d12", + "ratio11.25" + ] + }, + "config_fingerprint": "3d2656cfe49b6048", + "artifact_path": "experiments/thinkcleaned-d12-1ep-sh26-r11" + }, + "stage": "base", + "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "3d2656cfe49b6048" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 5, + "pos": 24369538, + "epoch": 1, + "pq_idx": 5, + "rg_idx": 24369538 + }, + "loop_state": { + "min_val_bpb": 1.2404084951472125, + "smooth_train_loss": 3.619971321170683, + "total_training_time": 2696.2369639873505, + "stage_training_flops": 465094777503744000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 465094777503744000 + } +} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_001500.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_001500.json new file mode 100644 index 0000000000000000000000000000000000000000..8246986f3f8fdf924d4705e976382fa5c513442b --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_001500.json @@ -0,0 +1,139 @@ +{ + "step": 1500, + "training_complete": false, + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "val_bpb": 1.185795900077561, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "thinkcleaned-d12-1ep-sh26-r11", + "wandb_run_id": "e8375954", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset-cleaned,d12,ratio11.25", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": 500, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints", + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json", + "tokenizer_fingerprint": "0a9922b59b5cb78a", + "git_commit_sha": "7f4c957646ab9d9314d7a7aae4f7efd73aa1ff2e", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "thinkcleaned-d12-1ep-sh26-r11", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset-clean", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 26, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "seed": 42, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "thinkcleaned-d12-1ep-sh26-r11", + "group": "think-d12", + "tags": [ + "think-dataset-cleaned", + "d12", + "ratio11.25" + ] + }, + "config_fingerprint": "3d2656cfe49b6048", + "artifact_path": "experiments/thinkcleaned-d12-1ep-sh26-r11" + }, + "stage": "base", + "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "3d2656cfe49b6048" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 7, + "pos": 86521538, + "epoch": 1, + "pq_idx": 7, + "rg_idx": 86521538 + }, + "loop_state": { + "min_val_bpb": 1.185795900077561, + "smooth_train_loss": 3.3830953942307014, + "total_training_time": 4026.685672521591, + "stage_training_flops": 697642166255616000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 697642166255616000 + } +} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_002000.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_002000.json new file mode 100644 index 0000000000000000000000000000000000000000..79434ae2af40ba1dd77a5ac64cdf17230d30a5a0 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_002000.json @@ -0,0 +1,139 @@ +{ + "step": 2000, + "training_complete": false, + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "val_bpb": 1.138428837130431, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "thinkcleaned-d12-1ep-sh26-r11", + "wandb_run_id": "e8375954", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset-cleaned,d12,ratio11.25", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": 500, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints", + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json", + "tokenizer_fingerprint": "0a9922b59b5cb78a", + "git_commit_sha": "7f4c957646ab9d9314d7a7aae4f7efd73aa1ff2e", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "thinkcleaned-d12-1ep-sh26-r11", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset-clean", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 26, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "seed": 42, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "thinkcleaned-d12-1ep-sh26-r11", + "group": "think-d12", + "tags": [ + "think-dataset-cleaned", + "d12", + "ratio11.25" + ] + }, + "config_fingerprint": "3d2656cfe49b6048", + "artifact_path": "experiments/thinkcleaned-d12-1ep-sh26-r11" + }, + "stage": "base", + "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "3d2656cfe49b6048" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 10, + "pos": 48673538, + "epoch": 1, + "pq_idx": 10, + "rg_idx": 48673538 + }, + "loop_state": { + "min_val_bpb": 1.138428837130431, + "smooth_train_loss": 3.5114756844749833, + "total_training_time": 5358.042316198349, + "stage_training_flops": 930189555007488000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 930189555007488000 + } +} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_002362.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_002362.json new file mode 100644 index 0000000000000000000000000000000000000000..c9051e61a453ef5eb8a7038a04187b32ad2b9355 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/meta_002362.json @@ -0,0 +1,139 @@ +{ + "step": 2362, + "training_complete": true, + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "val_bpb": 1.115457145344029, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "thinkcleaned-d12-1ep-sh26-r11", + "wandb_run_id": "e8375954", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset-cleaned,d12,ratio11.25", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": 500, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints", + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "experiment_config": "/content/nanochat_cache/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json", + "tokenizer_fingerprint": "0a9922b59b5cb78a", + "git_commit_sha": "7f4c957646ab9d9314d7a7aae4f7efd73aa1ff2e", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "thinkcleaned-d12-1ep-sh26-r11", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset-clean", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 26, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "seed": 42, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "thinkcleaned-d12-1ep-sh26-r11", + "group": "think-d12", + "tags": [ + "think-dataset-cleaned", + "d12", + "ratio11.25" + ] + }, + "config_fingerprint": "3d2656cfe49b6048", + "artifact_path": "experiments/thinkcleaned-d12-1ep-sh26-r11" + }, + "stage": "base", + "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "3d2656cfe49b6048" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 12, + "pos": 38471586, + "epoch": 1, + "pq_idx": 12, + "rg_idx": 38471586 + }, + "loop_state": { + "min_val_bpb": 1.115457145344029, + "smooth_train_loss": 3.2647400994218767, + "total_training_time": 6322.441261768341, + "stage_training_flops": 1098553864463843328, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1098553864463843328 + } +} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_000500.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_000500.pt new file mode 100644 index 0000000000000000000000000000000000000000..99f85fd3e3bb63d78b85d713ba26da167bdf5de5 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_000500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4b705d047f3a0c23bbc991ace9fb0415b75f81a419e74ad186324eebc49da397 +size 792761690 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_001000.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_001000.pt new file mode 100644 index 0000000000000000000000000000000000000000..b2d891898605a82b097c7a7f7831a488dad9ae2f --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_001000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:df783d2af59fb1d4bf290019d1c9529896b0f9a623d6df9a5859d63535d47b5b +size 792761690 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_001500.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_001500.pt new file mode 100644 index 0000000000000000000000000000000000000000..8544c8a7d77099209fb43e62441c1f758ed9bd87 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_001500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:837826c2b175cbfc616195d60418500377ffeaf649c2ec5fe3393bdc5bc22ca8 +size 792761690 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_002000.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_002000.pt new file mode 100644 index 0000000000000000000000000000000000000000..d36f4acefbb6d306f75e582e1b13cdb0544304ad --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_002000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f585203c4ba255265381a9e7d757dcbceb30114221054f9cc310334292d652c7 +size 792761690 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_002362.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_002362.pt new file mode 100644 index 0000000000000000000000000000000000000000..c0d59f7b444c6baf653cbfc6e0279b6897c9da60 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/model_002362.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:68d378385029fb2d10e84db5808dd6d9ee8bf7e807b17ab4a47e70e8965542bf +size 792761690 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_000500_rank0.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_000500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..675741712e9c45665e752ee96a1eb2ae5e3b715b --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_000500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c3ede0f1c2bdbaf19c446960c445a1c645e35a1aa476749d8d279f3fa825c43e +size 1246165357 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_001000_rank0.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_001000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..f3421bd84b57a7548a6bacf45bd8fd26c6e84029 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_001000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2607a60fcb952467cd994577a1ecbf2a48701154696d18b24333e9e0a126b866 +size 1246165357 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_001500_rank0.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_001500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..21891007cbd9d616966fdbf468b9bedafde5c3cd --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_001500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9872426092acb81cedaccd3527ad5f3b6e864e947bfb0fc13589fea32514658c +size 1246165357 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_002000_rank0.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_002000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..5f17de6bfeab68b303b4e39877e0125a5fdf9bed --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_002000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dea53faac8f5839609bc912f9ce25d64a824d2ea7c5bd922f98ff36c2b8950cf +size 1246165357 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_002362_rank0.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_002362_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..cfc5566f3ec8219e377453b5802fd3f7d8ddf398 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/base_checkpoints/optim_002362_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e146cae0ec7e4305313eba9e099afc97e789e9c2ba02613ee5a96f769da40a80 +size 1246165357 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json new file mode 100644 index 0000000000000000000000000000000000000000..f0f840d2413605a8e126c20033b0f778831b4008 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/config.json @@ -0,0 +1,56 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset-clean", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 26, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "seed": 42, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "thinkcleaned-d12-1ep-sh26-r11", + "group": "think-d12", + "tags": [ + "think-dataset-cleaned", + "d12", + "ratio11.25" + ] + }, + "config_fingerprint": "3d2656cfe49b6048", + "artifact_path": "experiments/thinkcleaned-d12-1ep-sh26-r11" +} diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb.json new file mode 100644 index 0000000000000000000000000000000000000000..068f75caea1d46adc479ace11d57ead0c66c2f8d --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb.json @@ -0,0 +1,12 @@ +{ + "model": "base_model (step 2362)", + "step": 2362, + "bpb": { + "val": 1.0608529946072547 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb_on_think-dataset-clean-1930s.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb_on_think-dataset-clean-1930s.json new file mode 100644 index 0000000000000000000000000000000000000000..be4f020cf240d19e781acc4beeee7d8dd9e28cc2 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb_on_think-dataset-clean-1930s.json @@ -0,0 +1,12 @@ +{ + "model": "base_model (step 2362)", + "step": 2362, + "bpb": { + "val": 1.0573472245939946 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb_on_think-dataset.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb_on_think-dataset.json new file mode 100644 index 0000000000000000000000000000000000000000..c98694dc87f14ca6025b83de7f86f21f7005f7cf --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/evals/val_bpb_on_think-dataset.json @@ -0,0 +1,12 @@ +{ + "model": "base_model (step 2362)", + "step": 2362, + "bpb": { + "val": 1.1045456977914099 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/run.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/run.json new file mode 100644 index 0000000000000000000000000000000000000000..73b780dbc0f5ae40b1c2932e3b54ac08a09e8367 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/run.json @@ -0,0 +1,10 @@ +{ + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "stage": "base", + "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "3d2656cfe49b6048", + "wandb_run_id": "e8375954", + "created_at": 1783267465 +} diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/summary.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/summary.json new file mode 100644 index 0000000000000000000000000000000000000000..bb6cc23b37f533f308934702fd5dd08ce8456b3c --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/summary.json @@ -0,0 +1,32 @@ +{ + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "stage": "base", + "base_experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "dataset": "jbduran/think-dataset-clean", + "dataset_revision": "main", + "step": 2362, + "depth": 12, + "target_param_data_ratio": 11.25, + "training_tokens": 1238368256, + "final_sampled_val_bpb": 1.115457145344029, + "minimum_sampled_val_bpb": 1.115457145344029, + "full_val_bpb": 1.0608529946072547, + "core_metric": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [], + "training_time_seconds": 6322.441261768341, + "stage_training_flops": 1.0985538644638433e+18, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1.0985538644638433e+18, + "config_fingerprint": "3d2656cfe49b6048", + "git_commit_sha": "7f4c957646ab9d9314d7a7aae4f7efd73aa1ff2e", + "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/e8375954", + "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/thinkcleaned-d12-1ep-sh26-r11", + "dataset_fingerprint": "b839fe2c141e60dd", + "tokenizer_fingerprint": "0a9922b59b5cb78a", + "unique_train_tokens": 1275519304, + "effective_epochs": 0.970873786164196 +} diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/experiment_tokenizer.json b/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/experiment_tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..973aef9df6a827762c0983a8a72ca075b0138ee5 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/experiment_tokenizer.json @@ -0,0 +1,18 @@ +{ + "experiment_id": "thinkcleaned-d12-1ep-sh26-r11", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset-clean", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 26, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "created_at": 1783267503 +} diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/token_bytes.pt b/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/token_bytes.pt new file mode 100644 index 0000000000000000000000000000000000000000..c9e8a074bb8475e3a950c6a4e610efd20658ef6b --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/token_bytes.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b24d3f9325f437305a97a56960479be10f222cc70fa50df4fe7f7809b3c51ec8 +size 132649 diff --git a/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/tokenizer.pkl b/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/tokenizer.pkl new file mode 100644 index 0000000000000000000000000000000000000000..5975ec52e31c1ab60dd365c96a500f5fb02ae409 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh26-r11/tokenizer/tokenizer.pkl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3dc059e668949894537dd6e813704ebb69db930d54de7181d6202f1d60f91636 +size 407574 diff --git a/experiments/thinkcleaned-d12-1ep-sh28-r11/config.json b/experiments/thinkcleaned-d12-1ep-sh28-r11/config.json new file mode 100644 index 0000000000000000000000000000000000000000..09926ff3799da9ec6b6551f53d914ade84f4544f --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh28-r11/config.json @@ -0,0 +1,56 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "thinkcleaned-d12-1ep-sh28-r11", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset-cleaned", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 24, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "seed": 42, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "thinkcleaned-d12-1ep-sh28-r11", + "group": "think-d12", + "tags": [ + "think-dataset-cleaned", + "d12", + "ratio11.25" + ] + }, + "config_fingerprint": "6d042a15b4b42e5e", + "artifact_path": "experiments/thinkcleaned-d12-1ep-sh28-r11" +} diff --git a/experiments/thinkcleaned-d12-1ep-sh28-r11/run.json b/experiments/thinkcleaned-d12-1ep-sh28-r11/run.json new file mode 100644 index 0000000000000000000000000000000000000000..d70e9e04c3dc4d206f3aa976ba058919155ddb74 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh28-r11/run.json @@ -0,0 +1,10 @@ +{ + "experiment_id": "thinkcleaned-d12-1ep-sh28-r11", + "stage": "base", + "base_experiment_id": "thinkcleaned-d12-1ep-sh28-r11", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "6d042a15b4b42e5e", + "wandb_run_id": "ac903c89", + "created_at": 1783266646 +} diff --git a/experiments/thinkcleaned-d12-1ep-sh29-r11/config.json b/experiments/thinkcleaned-d12-1ep-sh29-r11/config.json new file mode 100644 index 0000000000000000000000000000000000000000..e586213c3001162e411f71248bfc345fd5f13128 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh29-r11/config.json @@ -0,0 +1,56 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "thinkcleaned-d12-1ep-sh29-r11", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset-clean", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 24, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "seed": 42, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "thinkcleaned-d12-1ep-sh29-r11", + "group": "think-d12", + "tags": [ + "think-dataset-cleaned", + "d12", + "ratio11.25" + ] + }, + "config_fingerprint": "4509e64262826e28", + "artifact_path": "experiments/thinkcleaned-d12-1ep-sh29-r11" +} diff --git a/experiments/thinkcleaned-d12-1ep-sh29-r11/run.json b/experiments/thinkcleaned-d12-1ep-sh29-r11/run.json new file mode 100644 index 0000000000000000000000000000000000000000..814f824b061f4dd6510a3a23da50a4cde9b07130 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh29-r11/run.json @@ -0,0 +1,10 @@ +{ + "experiment_id": "thinkcleaned-d12-1ep-sh29-r11", + "stage": "base", + "base_experiment_id": "thinkcleaned-d12-1ep-sh29-r11", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "4509e64262826e28", + "wandb_run_id": "ba5e5413", + "created_at": 1783266942 +} diff --git a/experiments/thinkcleaned-d12-1ep-sh29-r11/tokenizer/experiment_tokenizer.json b/experiments/thinkcleaned-d12-1ep-sh29-r11/tokenizer/experiment_tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..fb61a5b6c9a2427345ffb5379810da50f3e16c58 --- /dev/null +++ b/experiments/thinkcleaned-d12-1ep-sh29-r11/tokenizer/experiment_tokenizer.json @@ -0,0 +1,18 @@ +{ + "experiment_id": "thinkcleaned-d12-1ep-sh29-r11", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset-clean", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 24, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "created_at": 1783266959 +}