"""Shared configuration. The two models differ ONLY in attention (bidirectional vs causal) and objective; everything else — size, data, tokenizer, sequence budget — is held equal so the comparison isolates the substrate (RQ3 / EXPERIMENTS.md).""" from __future__ import annotations from dataclasses import dataclass @dataclass class ModelConfig: vocab_size: int = 0 # filled from tokenizer d_model: int = 256 n_layers: int = 6 n_heads: int = 8 d_ff: int = 1024 dropout: float = 0.1 max_len: int = 512 # total sequence length T (shared) @dataclass class TaskConfig: # The whole source must fit the context (the masked block is filled in-place), # so T is set by source length. Attention is O(T^2) and MPS is sensitive to it; # T=512 (fp16) is the speed/coverage balance: fits most of difficulty 0-2, much # of 3; the long difficulty-4 programs are skipped (documented, not silently). seq_len: int = 512 # T: total canvas length mask_frac: float = 0.3 # default eval block fraction of the BODY frac_lo: float = 0.1 # training: sample block fraction in [lo, hi] per record frac_hi: float = 0.6 # so one model handles any masking ratio (eval sweeps it) tile_size: int = 16 # chars per tile: granularity of remask/revision max_decode: int = 256 # AR decode cap (>= largest block) block_len: int = 32 # block diffusion: tokens generated per block (KV-cached) n_inner: int = 4 # block diffusion: ReMDM refinement steps per block @dataclass class TrainConfig: batch_size: int = 64 steps: int = 4000 lr: float = 2.0e-4 # WSD peak (TRAINING.md) warmup: int = 200 weight_decay: float = 0.1 grad_clip: float = 1.0 log_every: int = 100 seed: int = 0