{ "smoke": false, "cuda": false, "backend": "cpu-host", "architecture": "gpt2-scratch", "param_count": 13867008, "n_layer": 6, "n_embd": 384, "n_head": 6, "vocab_size": 8192, "max_seq_len": 192, "steps": 180, "rows": 3175, "dataset_version": "1.4.0", "final_loss": 4.416327476501465, "losses_tail": [ 5.1331095695495605, 5.0547776222229, 4.5566725730896, 5.206869602203369, 6.316340923309326, 5.287867069244385, 5.436374664306641, 5.335423946380615, 5.663321495056152, 4.416327476501465 ], "hub_id_if_uploaded": "theworker02/open-reason-medium", "card_title": "Open Reason open-reason-medium (CPU)", "size_note": "This is a **medium** GPT-2-style causal LM trained from scratch on the Open Reason SFT split. It is larger than `theworker02/open-reason-small` and is **not** a 1B model and is **not** `theworker02/open-reason-1b`.", "note": "CPU causal LM. Not open-reason-1b. Not AMD GPU. No Reddit." }