| { | |
| "run_name": "model1_250m", | |
| "results_tag": "canonical", | |
| "hf_repo_id": "FAIRC/token-averaging-model1_250m", | |
| "checkpoint_format": { | |
| "keys": [ | |
| "step", | |
| "tokens_seen", | |
| "cumulative_flops", | |
| "model", | |
| "optimizer", | |
| "scheduler" | |
| ], | |
| "note": "Raw torch.save dict from experiments/chinchilla/train.py. Load with torch.load(..., map_location='cpu', weights_only=False) and take state['model']. Not a transformers AutoModel checkpoint." | |
| }, | |
| "model_config": { | |
| "name": "model1_250m", | |
| "d_model": 1024, | |
| "n_heads": 16, | |
| "n_layers": 16, | |
| "context_len": 1024, | |
| "averaging_k": 1, | |
| "method_name": null, | |
| "multi_token_phase_ratio": 0.0, | |
| "tie_embeddings": true, | |
| "grad_checkpoint": true, | |
| "color": "#4e9de0", | |
| "label": "~250M standard (n=1024)", | |
| "target_tokens": 5000000000, | |
| "lr": 0.00014, | |
| "warmup_steps": 2000, | |
| "n_params_approx": 252789760 | |
| } | |
| } | |