| { | |
| "checkpoint_id": "ckpt_20260422_013641_11000_22deff1b_9470fbb7", | |
| "created_at": "2026-04-22T01:36:41.599523", | |
| "iteration": 11000, | |
| "epoch": 0, | |
| "train_loss": 0.0, | |
| "val_loss": 0.00041066980917094044, | |
| "learning_rate": 0.00013425266109088577, | |
| "model_config": { | |
| "n_layer": 4, | |
| "n_head": 4, | |
| "n_embd": 256, | |
| "vocab_size": 50257, | |
| "block_size": 1024, | |
| "dropout": 0.1, | |
| "bias": true, | |
| "initial_connections": 0.1, | |
| "connection_growth_rate": 0.05, | |
| "max_connections": 1.0 | |
| }, | |
| "training_config": { | |
| "learning_rate": 0.0002, | |
| "batch_size": 2, | |
| "max_iters": 500, | |
| "warmup_iters": 5000, | |
| "lr_decay_iters": 50000, | |
| "min_lr": 1e-05, | |
| "weight_decay": 0.1, | |
| "grad_clip": 1.0, | |
| "enable_curriculum_learning": true, | |
| "enable_introspection": true | |
| }, | |
| "data_config": { | |
| "data_dir": "data/nanecho", | |
| "batch_size": 2, | |
| "block_size": 1024 | |
| }, | |
| "metrics": { | |
| "val_loss": 0.00041066980917094044, | |
| "connection_ratio": 1.0, | |
| "tokens_processed": 22528000, | |
| "training_speed_iters_per_sec": 0.10936917350797425 | |
| }, | |
| "tags": [ | |
| "phase_adaptive_mastery", | |
| "high_quality", | |
| "nanecho", | |
| "curriculum", | |
| "introspection" | |
| ], | |
| "parent_checkpoint": null, | |
| "notes": "Training checkpoint at iteration 11000 (resumed from iteration 10500) | Phase: adaptive_mastery", | |
| "file_size_mb": 253.3669786453247, | |
| "quality_score": 1689600.7829461372 | |
| } |