{ "run_name": "model1_500m", "results_tag": "canonical", "hf_repo_id": "FAIRC/token-averaging-model1_500m", "checkpoint_format": { "keys": [ "step", "tokens_seen", "cumulative_flops", "model", "optimizer", "scheduler" ], "note": "Raw torch.save dict from experiments/chinchilla/train.py. Load with torch.load(..., map_location='cpu', weights_only=False) and take state['model']. Not a transformers AutoModel checkpoint." }, "model_config": { "name": "model1_500m", "d_model": 1280, "n_heads": 20, "n_layers": 22, "context_len": 1024, "averaging_k": 1, "method_name": null, "multi_token_phase_ratio": 0.0, "tie_embeddings": true, "grad_checkpoint": false, "color": "#4e9de0", "label": "~500M standard (n=1024)", "target_tokens": 10000000000, "lr": 0.00012, "warmup_steps": 2000, "n_params_approx": 496866560 } }