{ "run_name": "model1_250m", "results_tag": "canonical", "hf_repo_id": "FAIRC/token-averaging-model1_250m", "checkpoint_format": { "keys": [ "step", "tokens_seen", "cumulative_flops", "model", "optimizer", "scheduler" ], "note": "Raw torch.save dict from experiments/chinchilla/train.py. Load with torch.load(..., map_location='cpu', weights_only=False) and take state['model']. Not a transformers AutoModel checkpoint." }, "model_config": { "name": "model1_250m", "d_model": 1024, "n_heads": 16, "n_layers": 16, "context_len": 1024, "averaging_k": 1, "method_name": null, "multi_token_phase_ratio": 0.0, "tie_embeddings": true, "grad_checkpoint": true, "color": "#4e9de0", "label": "~250M standard (n=1024)", "target_tokens": 5000000000, "lr": 0.00014, "warmup_steps": 2000, "n_params_approx": 252789760 } }