[ { "model_name": "Baseline", "batch_size": 32, "seq_len": 81, "hidden_size": 512, "param_count": 27296770, "latency_mean_ms": 12.367569541931152, "latency_std_ms": 0.03661433936324098, "throughput_sps": 2587.4121743570417, "gpu_memory_mb": 174.16943359375 }, { "model_name": "Tiered", "batch_size": 32, "seq_len": 81, "hidden_size": 512, "param_count": 27296770, "latency_mean_ms": 11.73528323173523, "latency_std_ms": 0.06700267067328709, "throughput_sps": 2726.81957206314, "gpu_memory_mb": 193.54443359375 } ]