{ "d_model": 128, "nhead": 4, "num_layers": 4, "vocab_size": 257, "avg_loss": 0.05647050041671504, "avg_bpb": 0.08146971090771268 }