echo-1 / ml /config.py
lupodevelop's picture
echo-1 Stage 0 explainer: diffusion vs autoregressive, execution-verified
3afc977 verified
Raw
History Blame Contribute Delete
1.79 kB
"""Shared configuration. The two models differ ONLY in attention (bidirectional
vs causal) and objective; everything else — size, data, tokenizer, sequence
budget — is held equal so the comparison isolates the substrate (RQ3 / EXPERIMENTS.md)."""
from __future__ import annotations
from dataclasses import dataclass
@dataclass
class ModelConfig:
vocab_size: int = 0 # filled from tokenizer
d_model: int = 256
n_layers: int = 6
n_heads: int = 8
d_ff: int = 1024
dropout: float = 0.1
max_len: int = 512 # total sequence length T (shared)
@dataclass
class TaskConfig:
# The whole source must fit the context (the masked block is filled in-place),
# so T is set by source length. Attention is O(T^2) and MPS is sensitive to it;
# T=512 (fp16) is the speed/coverage balance: fits most of difficulty 0-2, much
# of 3; the long difficulty-4 programs are skipped (documented, not silently).
seq_len: int = 512 # T: total canvas length
mask_frac: float = 0.3 # default eval block fraction of the BODY
frac_lo: float = 0.1 # training: sample block fraction in [lo, hi] per record
frac_hi: float = 0.6 # so one model handles any masking ratio (eval sweeps it)
tile_size: int = 16 # chars per tile: granularity of remask/revision
max_decode: int = 256 # AR decode cap (>= largest block)
block_len: int = 32 # block diffusion: tokens generated per block (KV-cached)
n_inner: int = 4 # block diffusion: ReMDM refinement steps per block
@dataclass
class TrainConfig:
batch_size: int = 64
steps: int = 4000
lr: float = 2.0e-4 # WSD peak (TRAINING.md)
warmup: int = 200
weight_decay: float = 0.1
grad_clip: float = 1.0
log_every: int = 100
seed: int = 0