| { |
| "description": "Checkpoints for 'Temporal Conditioning for Human Behavioral Cloning in Chess' (IEEE GEM 2026).", |
| "note": "Parameter counts are unique parameters. A naive sum over state_dict() is 4,194,304 larger because the GAB weight is shared across all 8 blocks and counted 8 times.", |
| "checkpoints": [ |
| { |
| "file": "marvin_23m_cond_1M.pt", |
| "role": "23M time-conditioned, 1M steps. Headline arm of the controlled ablation (Tables I-III, Fig. 1).", |
| "source_filename": "chessformer_maia3-23m_elo-endpoints_time-tokens_tempotok_s67_step1000000.pt", |
| "size_bytes": 280134073, |
| "steps": 1000000, |
| "seed": 67, |
| "data_seed": 42, |
| "parameters": 23315934, |
| "time_conditioned": true, |
| "config_dict": { |
| "history": 8, |
| "num_blocks": 8, |
| "mlp_ratio": 2.0, |
| "dim_emb": 128, |
| "dropout": 0.0, |
| "use_gab": true, |
| "use_rms_norm": true, |
| "omit_qkv_biases": true, |
| "activation": "gelu", |
| "time_encoder": "tokens", |
| "time_token_encoder": "interp", |
| "time_featurize": "log1p", |
| "per_position_clocks": false, |
| "shrink_global_time": false, |
| "per_position_time_deltas": false, |
| "no_time_context": false, |
| "num_piece_types": 13, |
| "gradient_checkpointing": false, |
| "dim_vit": 512, |
| "head_hid_dim": 512, |
| "num_heads": 16, |
| "gab_gen_size": 128, |
| "gab_per_square_dim": 32, |
| "gab_intermediate_dim": 128, |
| "tempo_history_token": true, |
| "zero_time_history": false, |
| "elo_encoder": "endpoints" |
| }, |
| "train_command": "python train.py --config 23m --batch-size 256 --grad-accum-steps 4 --lr 1.25e-4 --seed 67 --data-seed 42 --data-dir data/train --val-data-dir data/val", |
| "sha256": "ee82bd936f492bb5cce97baf4335172bb7f6dca71a6652d7eedae0ed7bfcc57f" |
| }, |
| { |
| "file": "marvin_23m_blind_1M.pt", |
| "role": "23M time-blind twin, 1M steps. Identical backbone, data and schedule; receives zeroed temporal inputs. Training stopped here, so this is also the latest time-blind checkpoint.", |
| "source_filename": "chessformer_23m_elo-endpoints_time-blind_s67_step1000000.pt", |
| "size_bytes": 279261361, |
| "steps": 1000000, |
| "seed": 67, |
| "data_seed": 42, |
| "parameters": 23249117, |
| "time_conditioned": false, |
| "config_dict": { |
| "history": 8, |
| "num_blocks": 8, |
| "mlp_ratio": 2.0, |
| "dim_emb": 128, |
| "dropout": 0.0, |
| "use_gab": true, |
| "use_rms_norm": true, |
| "omit_qkv_biases": true, |
| "activation": "gelu", |
| "gab_additive_bias": true, |
| "gab_qk_rotation": false, |
| "time_encoder": "tokens", |
| "time_token_encoder": "interp", |
| "time_featurize": "log1p", |
| "tempo_history_token": true, |
| "per_position_clocks": false, |
| "shrink_global_time": false, |
| "per_position_time_deltas": false, |
| "no_time_context": false, |
| "num_piece_types": 13, |
| "gradient_checkpointing": false, |
| "dim_vit": 512, |
| "head_hid_dim": 512, |
| "num_heads": 16, |
| "gab_gen_size": 128, |
| "gab_per_square_dim": 32, |
| "gab_intermediate_dim": 128, |
| "zero_time_history": false, |
| "no_time_conditioning": true, |
| "elo_encoder": "endpoints" |
| }, |
| "train_command": "python train.py --config 23m --batch-size 256 --grad-accum-steps 4 --lr 1.25e-4 --seed 67 --data-seed 42 --data-dir data/train --val-data-dir data/val --no-time-conditioning", |
| "sha256": "b8b6d40132bb1f21e12cdc801bd674efe07de106a6cb36cc6eeecf1073c63f54" |
| }, |
| { |
| "file": "marvin_79m_cond_1M.pt", |
| "role": "79M time-conditioned at 1M steps, for step-matched comparison with the 23M pair.", |
| "source_filename": "chessformer_79m_elo-endpoints_time-tokens_tempotok_s67_step1000000.pt", |
| "size_bytes": 956000269, |
| "steps": 1000000, |
| "seed": 67, |
| "data_seed": 42, |
| "parameters": 79641566, |
| "time_conditioned": true, |
| "config_dict": { |
| "history": 8, |
| "num_blocks": 8, |
| "mlp_ratio": 2.0, |
| "dim_emb": 128, |
| "dropout": 0.0, |
| "use_gab": true, |
| "use_rms_norm": true, |
| "omit_qkv_biases": true, |
| "activation": "gelu", |
| "gab_additive_bias": true, |
| "gab_qk_rotation": false, |
| "time_encoder": "tokens", |
| "time_token_encoder": "interp", |
| "time_featurize": "log1p", |
| "tempo_history_token": true, |
| "per_position_clocks": false, |
| "shrink_global_time": false, |
| "per_position_time_deltas": false, |
| "no_time_context": false, |
| "num_piece_types": 13, |
| "gradient_checkpointing": false, |
| "dim_vit": 1024, |
| "head_hid_dim": 1024, |
| "num_heads": 32, |
| "gab_gen_size": 128, |
| "gab_per_square_dim": 32, |
| "gab_intermediate_dim": 128, |
| "zero_time_history": false, |
| "elo_encoder": "endpoints" |
| }, |
| "train_command": "python train.py --config 79m --batch-size 256 --grad-accum-steps 4 --lr 1e-4 --seed 67 --data-seed 42 --data-dir data/train --val-data-dir data/val", |
| "sha256": "8bb522116f1e9edf4c20a4292216addb0983e2232d9a6ff37a248330c73aca84" |
| }, |
| { |
| "file": "marvin_23m_cond_1700k.pt", |
| "role": "23M time-conditioned, 1.7M steps. Longest-trained 23M and the strongest by validation loss.", |
| "source_filename": "chessformer_maia3-23m_elo-endpoints_time-tokens_tempotok_s67_latest.pt", |
| "size_bytes": 280093471, |
| "steps": 1700000, |
| "seed": 67, |
| "data_seed": 42, |
| "parameters": 23315934, |
| "time_conditioned": true, |
| "config_dict": { |
| "history": 8, |
| "num_blocks": 8, |
| "mlp_ratio": 2.0, |
| "dim_emb": 128, |
| "dropout": 0.0, |
| "use_gab": true, |
| "use_rms_norm": true, |
| "omit_qkv_biases": true, |
| "activation": "gelu", |
| "time_encoder": "tokens", |
| "time_token_encoder": "interp", |
| "time_featurize": "log1p", |
| "per_position_clocks": false, |
| "shrink_global_time": false, |
| "per_position_time_deltas": false, |
| "no_time_context": false, |
| "num_piece_types": 13, |
| "gradient_checkpointing": false, |
| "dim_vit": 512, |
| "head_hid_dim": 512, |
| "num_heads": 16, |
| "gab_gen_size": 128, |
| "gab_per_square_dim": 32, |
| "gab_intermediate_dim": 128, |
| "tempo_history_token": true, |
| "zero_time_history": false, |
| "elo_encoder": "endpoints" |
| }, |
| "train_command": "python train.py --config 23m --batch-size 256 --grad-accum-steps 4 --lr 1.25e-4 --seed 67 --data-seed 42 --data-dir data/train --val-data-dir data/val", |
| "sha256": "10679ea7b74e0d924a31c65977f2b2563a305ca3ba159d891cd0251252698861" |
| }, |
| { |
| "file": "marvin_79m_cond_1650k.pt", |
| "role": "79M time-conditioned, 1.65M steps. The 79M model the paper evaluates against released Maia-3 (Fig. 2, Table IV).", |
| "source_filename": "chessformer_79m_elo-endpoints_time-tokens_tempotok_s67_latest.pt", |
| "size_bytes": 955983347, |
| "steps": 1650000, |
| "seed": 67, |
| "data_seed": 42, |
| "parameters": 79641566, |
| "time_conditioned": true, |
| "config_dict": { |
| "history": 8, |
| "num_blocks": 8, |
| "mlp_ratio": 2.0, |
| "dim_emb": 128, |
| "dropout": 0.0, |
| "use_gab": true, |
| "use_rms_norm": true, |
| "omit_qkv_biases": true, |
| "activation": "gelu", |
| "gab_additive_bias": true, |
| "gab_qk_rotation": false, |
| "time_encoder": "tokens", |
| "time_token_encoder": "interp", |
| "time_featurize": "log1p", |
| "tempo_history_token": true, |
| "per_position_clocks": false, |
| "shrink_global_time": false, |
| "per_position_time_deltas": false, |
| "no_time_context": false, |
| "num_piece_types": 13, |
| "gradient_checkpointing": false, |
| "dim_vit": 1024, |
| "head_hid_dim": 1024, |
| "num_heads": 32, |
| "gab_gen_size": 128, |
| "gab_per_square_dim": 32, |
| "gab_intermediate_dim": 128, |
| "zero_time_history": false, |
| "elo_encoder": "endpoints" |
| }, |
| "train_command": "python train.py --config 79m --batch-size 256 --grad-accum-steps 4 --lr 1e-4 --seed 67 --data-seed 42 --data-dir data/train --val-data-dir data/val", |
| "sha256": "047e914d0d71be607b95c2a93d2e255ab3c97ff1b7c63602f3c1064a00709628" |
| } |
| ] |
| } |