Spaces:
Sleeping
Sleeping
File size: 18,028 Bytes
0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 3c35b12 0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 3c35b12 0623ff0 3c35b12 0623ff0 3c35b12 0623ff0 3c35b12 0623ff0 3c35b12 0623ff0 3c35b12 0623ff0 3c35b12 0623ff0 3c35b12 0623ff0 3c35b12 0623ff0 a4508e8 0623ff0 a4508e8 3c35b12 0623ff0 3c35b12 0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 a4508e8 0623ff0 3c35b12 a4508e8 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 | """Feature engineering for GPU Perf Prophet: build_training_df(raw_df) filters/enriches/featurises raw MLPerf rows into a model-ready DataFrame; roofline_ceilings(...) is the pure-function roofline computation reused by the pipeline and notebooks."""
from __future__ import annotations
import logging
import re
from pathlib import Path
from typing import Optional
import pandas as pd
from src.data.gpu_spec_db import enrich_df
log = logging.getLogger(__name__)
# --- Reference tables ---
# benchmark_base → (total_params_b, compute_params_b): total = all weights in VRAM (bandwidth ceiling / VRAM-fit check); compute = active params per forward pass (dense: same as total; MoE: active-expert only) — e.g. Mixtral 8x7B approximated as total=46.7B (8 x ~7B experts), active=14.1B (2 active experts x 6.7B + ~0.7B shared layers); sources: Meta Llama 2/3 papers/blog, EleutherAI GPT-J-6B model card, Mistral AI blog (2023).
MODEL_PARAMS: dict[str, tuple[float, float]] = {
"llama2-70b": (70.0, 70.0),
"llama3.1-405b": (405.0, 405.0),
"llama3.1-8b": (8.03, 8.03), # anticipated for MLPerf v7.0+; no rows yet
"gptj": (6.05, 6.05),
"mixtral-8x7b": (46.7, 14.1), # (total, active)
}
# benchmark_base → (n_layers, n_kv_heads, head_dim) for KV-cache sizing, not derivable from MLPerf rows — sourced from each model's published config.json/paper (GQA models have a proportionally smaller KV cache than MHA); Mixtral 8x7B's MoE only changes the FFN, so its KV-cache math is treated as a dense 32-layer/8-kv-head/128-head-dim model with no sliding window (unlike base Mistral 7B v0.1).
MODEL_ARCH: dict[str, tuple[int, int, int]] = {
"llama2-70b": (80, 8, 128),
"llama3.1-405b": (126, 8, 128),
"llama3.1-8b": (32, 8, 128),
"gptj": (28, 16, 256),
"mixtral-8x7b": (32, 8, 128),
}
# benchmark_accuracy_tier → precision label used to select peak TFLOPS and bytes-per-param: "99.9"→FP16 (near-lossless), "99"→FP8 (modest accuracy drop, halves memory), "base"→BF16 (loosest constraint); if a GPU lacks the selected precision, _select_peak_tflops falls back to FP16 for TRAINING-DATA ingestion only — the live serving path must raise an "unsupported precision" error instead, see gpu_supports_precision() below.
TIER_TO_PRECISION: dict[str, str] = {
"99.9": "fp16",
"99": "fp8",
"base": "bf16",
}
# Bytes occupied per stored parameter at each precision.
BYTES_PER_PARAM: dict[str, float] = {
"fp32": 4.0,
"bf16": 2.0,
"fp16": 2.0,
"fp8": 1.0,
"fp6": 0.75,
"fp4": 0.5,
"int8": 1.0,
}
# GPU peak TFLOPS column name for each precision label.
_PRECISION_TO_COL: dict[str, str] = {
"fp32": "gpu_peak_fp32_tflops",
"bf16": "gpu_peak_bf16_tflops",
"fp16": "gpu_peak_fp16_tflops",
"fp8": "gpu_peak_fp8_tflops",
"fp6": "gpu_peak_fp6_tflops",
"fp4": "gpu_peak_fp4_tflops",
"int8": "gpu_peak_int8_tops",
}
# Framework string → normalized family label, matched in order (first hit wins).
_FRAMEWORK_PATTERNS: list[tuple[re.Pattern, str]] = [
(re.compile(r"TensorRT", re.IGNORECASE), "tensorrt"),
(re.compile(r"vLLM", re.IGNORECASE), "vllm"),
(re.compile(r"ROCm|Mango", re.IGNORECASE), "rocm_other"),
]
# Per-vendor architecture generation ordinals (higher = newer), split by vendor so the model can't learn spurious cross-vendor "newer = better" patterns (e.g. CDNA3 > Hopper is meaningless) — NaN for the other vendor's GPUs is intentional.
_NVIDIA_ARCH_ORDINAL: dict[str, int] = {
"ampere": 1,
"ada_lovelace": 2,
"hopper": 3,
"blackwell": 4,
}
_AMD_ARCH_ORDINAL: dict[str, int] = {
"cdna3": 1,
"cdna4": 2,
}
# MLPerf round tag → chronological ordinal (higher = more recent), prices in framework/driver maturity (e.g. ROCm version) as a feature since early rounds run on less-tuned software stacks the model could otherwise mistake for a hardware effect; unrecognized round tags map to NaN rather than raising, same convention as the arch ordinals above.
ROUND_ORDINAL: dict[str, int] = {
"v4.1": 1,
"v5.0": 2,
"v5.1": 3,
"v6.0": 4,
}
# --- Core roofline computation (pure function — also used by notebooks directly) ---
def roofline_ceilings(
total_params_b: float,
compute_params_b: float,
bytes_per_param: float,
hbm_bw_tbps: float,
peak_tflops: float,
) -> tuple[float, float, float]:
"""Return (bandwidth_ceiling, compute_ceiling, roofline_tput) in tokens/sec: bandwidth_ceiling is a memory-bandwidth-richness proxy using *total* params (VRAM footprint, so MoE reflects HBM occupancy not per-token access), compute_ceiling is the hard physical throughput ceiling using *compute* (active) params, and roofline_tput = compute_ceiling (the correct upper bound for batched LLM inference; bandwidth_ceiling is kept as a separate feature rather than folded in)."""
model_bytes = total_params_b * 1e9 * bytes_per_param # bytes
bw_bytes_per_sec = hbm_bw_tbps * 1e12 # bytes/s
bw_ceil = bw_bytes_per_sec / model_bytes # tokens/s
flops_per_token = 2.0 * compute_params_b * 1e9 # FLOPs
peak_flops_per_sec = peak_tflops * 1e12 # FLOPs/s
compute_ceil = peak_flops_per_sec / flops_per_token # tokens/s
return bw_ceil, compute_ceil, compute_ceil
# Default batch/context-length assumption for KV-cache sizing — MLPerf rows carry no per-row batch/context length so this can't be learned like efficiency_ratio; it's a stated, overridable-per-request assumption (same spirit as the static pricing snapshot) representing a moderately loaded Offline-scenario batched-serving workload.
DEFAULT_BATCH_SIZE: int = 32
DEFAULT_INPUT_TOKENS: int = 2048
DEFAULT_OUTPUT_TOKENS: int = 256
MIN_BATCH_SIZE, MAX_BATCH_SIZE = 1, 256
MIN_INPUT_TOKENS, MAX_INPUT_TOKENS = 64, 8192
MIN_OUTPUT_TOKENS, MAX_OUTPUT_TOKENS = 1, 4096
# Outlier-rejection bound on efficiency_ratio: a row outside (0, MAX_EFFICIENCY_RATIO] signals a spec-DB or parse error (e.g. the precision-proxy mismatch that drove pre-FP8-override AMD 99.9-tier rows to ~1.35) and is dropped; values in (1.0, 1.2] are still kept as expected precision-proxy noise, see the diagnostic-only warning in build_training_df.
MAX_EFFICIENCY_RATIO: float = 1.2
# 10% activation/framework overhead on top of weights + KV cache, aligned with vLLM's default --gpu-memory-utilization 0.90.
MEMORY_OVERHEAD_FACTOR: float = 1.10
# Verdict thresholds on VRAM utilization: does_not_fit is a hard exclusion, tight is a disclosure-only flag (expected to run but with little headroom for allocator fragmentation).
_FITS_MAX_UTIL: float = 0.90
_TIGHT_MAX_UTIL: float = 0.98
# The only values memory_fit_verdict() ever returns — single source of truth for closed-set membership checks (mirrors VALID_FRAMEWORKS/_normalize_framework).
VALID_MEMORY_FIT_VERDICTS: frozenset[str] = frozenset({"fits", "tight", "does_not_fit"})
def validate_serving_shape(batch_size: int, input_tokens: int, output_tokens: int) -> None:
"""Raise ValueError if batch_size/input_tokens/output_tokens are out of range; single source of truth for this check shared by GpuPredictor.predict()/predict_batch() and GpuRecommender.recommend() so both entry points enforce the same input contract (recommend() previously had no check at all)."""
if not (MIN_BATCH_SIZE <= batch_size <= MAX_BATCH_SIZE):
raise ValueError(
f"Invalid batch_size {batch_size!r}. "
f"Valid range: [{MIN_BATCH_SIZE}, {MAX_BATCH_SIZE}]"
)
if not (MIN_INPUT_TOKENS <= input_tokens <= MAX_INPUT_TOKENS):
raise ValueError(
f"Invalid input_tokens {input_tokens!r}. "
f"Valid range: [{MIN_INPUT_TOKENS}, {MAX_INPUT_TOKENS}]"
)
if not (MIN_OUTPUT_TOKENS <= output_tokens <= MAX_OUTPUT_TOKENS):
raise ValueError(
f"Invalid output_tokens {output_tokens!r}. "
f"Valid range: [{MIN_OUTPUT_TOKENS}, {MAX_OUTPUT_TOKENS}]"
)
def kv_cache_gb(
n_layers: int,
n_kv_heads: int,
head_dim: int,
batch_size: int,
input_tokens: int,
output_tokens: int,
bytes_per_value: float,
) -> float:
"""KV-cache size in GB for one batch at the given context length: 2 (K and V) x batch x seq_len x n_layers x n_kv_heads x head_dim x bytes; GQA models (n_kv_heads < n_heads) shrink this proportionally, the reduction that makes GQA cheap to serve."""
seq_len = input_tokens + output_tokens
kv_bytes = (
2 * batch_size * seq_len * n_layers * n_kv_heads * head_dim * bytes_per_value
)
return kv_bytes / 1e9
def memory_fit_verdict(
weights_gb: float,
kv_gb: float,
vram_gb: float,
) -> tuple[str, float, float]:
"""Return (verdict, total_gb, utilization), where verdict is one of "fits" (util <= 0.90), "tight" (<= 0.98), or "does_not_fit" (> 0.98)."""
total_gb = (weights_gb + kv_gb) * MEMORY_OVERHEAD_FACTOR
utilization = total_gb / vram_gb
if utilization <= _FITS_MAX_UTIL:
verdict = "fits"
elif utilization <= _TIGHT_MAX_UTIL:
verdict = "tight"
else:
verdict = "does_not_fit"
return verdict, total_gb, utilization
def cost_per_million_tokens(
price_per_gpu_hr: Optional[float],
tokens_per_sec: float,
) -> Optional[float]:
"""USD per 1M tokens served: (usd_per_hour / 3600) / (tok/s / 1e6); returns None when price is unknown or throughput is non-positive (undefined, not zero or an error, matching cost_efficiency's None-for-unpriced convention)."""
if price_per_gpu_hr is None or tokens_per_sec <= 0:
return None
return (price_per_gpu_hr / 3600.0) / (tokens_per_sec / 1_000_000.0)
# --- Internal helpers ---
def _normalize_framework(raw: Optional[str]) -> str:
if not isinstance(raw, str):
return "unknown"
for pattern, label in _FRAMEWORK_PATTERNS:
if pattern.search(raw):
return label
return "other"
def _select_peak_tflops(row: pd.Series, precision: str) -> Optional[float]:
"""Return the peak TFLOPS for `precision`, falling back to fp16 if absent — training-data ingestion only, see the note on gpu_supports_precision()."""
col = _PRECISION_TO_COL.get(precision)
val = row.get(col) if col else None
if val is None or (isinstance(val, float) and pd.isna(val)):
# GPU doesn't support this precision natively — fall back to fp16.
val = row.get("gpu_peak_fp16_tflops")
return val
def gpu_supports_precision(gpu_spec: dict, precision: str) -> bool:
"""Whether gpu_spec's peak_tflops table has a real (non-null) entry for `precision` — single source of truth for the rule that unsupported precision must raise rather than silently substitute; `gpu_specs.yaml` encodes non-support as `~` (e.g. `a100_sxm_80gb.peak_tflops.fp8: ~`, Ampere has no native FP8 path), and both GpuPredictor and GpuRecommender call this before building a prediction."""
peak = (gpu_spec.get("peak_tflops") or {}).get(precision)
if peak is None:
return False
if isinstance(peak, float) and pd.isna(peak):
return False
return True
# --- Public pipeline ---
def build_training_df(
raw_df: pd.DataFrame,
spec_path: Optional[Path] = None,
) -> pd.DataFrame:
"""Filter to result_valid rows, enrich with GPU specs, attach model-param references, compute roofline ceilings, and derive secondary features (efficiency ratio, VRAM fit, framework family, architecture ordinal, vendor indicator); rows that can't be featurised (unknown benchmark_base or missing GPU specs) are dropped with a warning rather than propagating NaN into training."""
kwargs = {"spec_path": spec_path} if spec_path is not None else {}
df = enrich_df(raw_df, **kwargs).copy()
# --- filter --- gpu_in_model_scope gates recommendation exposure, not training inclusion; out-of-scope GPUs (e.g. B200, H200 NVL) remain valid training signal even though not served to users v1.
df = df[df["result_valid"]]
log.info("After result_valid filter: %d rows", len(df))
# --- model params ---
df["model_total_params_b"] = df["benchmark_base"].map(
{k: v[0] for k, v in MODEL_PARAMS.items()}
)
df["model_compute_params_b"] = df["benchmark_base"].map(
{k: v[1] for k, v in MODEL_PARAMS.items()}
)
unknown_benchmarks = df[df["model_total_params_b"].isna()]["benchmark_base"].unique()
if len(unknown_benchmarks):
log.warning("Dropping %d rows with unknown benchmark_base: %s",
df["model_total_params_b"].isna().sum(), unknown_benchmarks)
df = df[df["model_total_params_b"].notna()]
# --- precision selection ---
df["selected_precision"] = df["benchmark_accuracy_tier"].map(TIER_TO_PRECISION)
# AMD CDNA hardware achieves 99.9 accuracy with FP8, not FP16 (the TIER_TO_PRECISION default, correct for NVIDIA) — override for AMD so efficiency_ratio is computed against the right (2x FP16 TFLOPS) ceiling, eliminating ceiling violations in training.
amd_tier_99_9 = (df["gpu_vendor"] == "amd") & (df["benchmark_accuracy_tier"] == "99.9")
df.loc[amd_tier_99_9, "selected_precision"] = "fp8"
df["bytes_per_param"] = df["selected_precision"].map(BYTES_PER_PARAM)
# Vectorised equivalent of _select_peak_tflops over all rows: covers the three TIER_TO_PRECISION values, falling back to fp16 for unknown precision or NaN selected-precision (same as the scalar helper).
_fp16 = df["gpu_peak_fp16_tflops"]
_fp8 = df["gpu_peak_fp8_tflops"].where(df["gpu_peak_fp8_tflops"].notna(), _fp16)
_bf16 = df["gpu_peak_bf16_tflops"].where(df["gpu_peak_bf16_tflops"].notna(), _fp16)
_prec = df["selected_precision"]
df["peak_tflops_selected"] = _fp8.where(
_prec == "fp8", _bf16.where(_prec == "bf16", _fp16)
)
# --- roofline ---
missing_specs = df[
df["gpu_hbm_bandwidth_tbps"].isna() | df["peak_tflops_selected"].isna()
]
if len(missing_specs):
log.warning("Dropping %d rows with missing GPU specs (cannot compute roofline)",
len(missing_specs))
df = df[
df["gpu_hbm_bandwidth_tbps"].notna() & df["peak_tflops_selected"].notna()
]
_model_bytes = df["model_total_params_b"] * 1e9 * df["bytes_per_param"]
df["bandwidth_ceiling_tok_per_sec"] = (df["gpu_hbm_bandwidth_tbps"] * 1e12) / _model_bytes
df["compute_ceiling_tok_per_sec"] = (df["peak_tflops_selected"] * 1e12) / (
2.0 * df["model_compute_params_b"] * 1e9
)
df["roofline_tput"] = df["compute_ceiling_tok_per_sec"]
# --- derived features ---
df["efficiency_ratio"] = (
df["throughput_tok_per_sec_per_gpu"] / df["roofline_tput"]
)
df["model_size_gb"] = (
df["model_total_params_b"] * df["bytes_per_param"]
)
df["model_to_vram_ratio"] = df["model_size_gb"] / df["gpu_vram_gb"]
# Deduplicate framework strings before normalizing (~10-20 unique values across ~1112 rows) to avoid ~1112 redundant _normalize_framework calls — parallels the per-unique-gpu-name optimization in enrich_df.
_fw_unique_map = {fw: _normalize_framework(fw) for fw in df["framework"].dropna().unique()}
df["framework_family"] = df["framework"].map(_fw_unique_map).fillna("unknown")
df["nvidia_arch_gen"] = df["gpu_architecture"].map(_NVIDIA_ARCH_ORDINAL)
df["amd_arch_gen"] = df["gpu_architecture"].map(_AMD_ARCH_ORDINAL)
df["vendor_is_amd"] = (df["gpu_vendor"] == "amd").astype(int)
df["mlperf_round_num"] = df["round"].map(ROUND_ORDINAL)
# Binary flag separating "base" tier (BF16) from "99"/"99.9" (FP8/FP16 or AMD FP8 override); replaces three-level accuracy_tier_ord, which became a spurious discriminator within AMD LOGO folds once the AMD FP8 override made "99" and "99.9" both FP8 — bytes_per_param already covers the FP8-vs-FP16 split, is_base_tier covers the remaining base-vs-non-base distinction it can't express (base BF16 = 2.0, same value as NVIDIA FP16).
df["is_base_tier"] = (df["benchmark_accuracy_tier"] == "base").astype(int)
# Efficiency ratio > 1 (actual throughput exceeds the inferred-precision compute ceiling) is expected for AMD CDNA4 "99.9"-tier rows, whose vLLM/ROCm stack really achieves 99.9 accuracy with FP8 while our proxy maps 99.9→FP16 — these rows are valid training data showing AMD outperforming the FP16 ceiling for this tier.
n_violations = (df["throughput_tok_per_sec_per_gpu"] > df["roofline_tput"]).sum()
if n_violations:
log.warning(
"%d rows (%.1f%%) have throughput > compute ceiling at selected "
"precision — likely precision-proxy mismatch (AMD FP8 at 99.9 tier).",
n_violations,
100 * n_violations / len(df),
)
# Outlier-rejection rule (hard bound, unlike the >1.0 warning above): drop rows whose efficiency_ratio falls outside (0, MAX_EFFICIENCY_RATIO], including NaN/inf from a zero or missing roofline_tput; .notna() is technically redundant with `> 0` (NaN comparisons are always False under IEEE 754, confirmed by mutation testing 2026-07-12) but kept explicit for readability.
valid_ratio = (
df["efficiency_ratio"].notna()
& (df["efficiency_ratio"] > 0)
& (df["efficiency_ratio"] <= MAX_EFFICIENCY_RATIO)
)
if (~valid_ratio).any():
dropped = df.loc[~valid_ratio, "efficiency_ratio"]
log.warning(
"Dropping %d rows with efficiency_ratio outside (0, %.1f] "
"(sample values: %s) — outlier-rejection rule; likely a "
"spec-DB or parse error, not valid training signal.",
len(dropped),
MAX_EFFICIENCY_RATIO,
sorted(dropped.round(3).tolist())[:10],
)
df = df[valid_ratio]
log.info("build_training_df complete: %d rows, %d columns", len(df), df.shape[1])
return df.reset_index(drop=True)
|