Spaces:
Sleeping
Sleeping
File size: 7,751 Bytes
0623ff0 a4508e8 3c35b12 a4508e8 3c35b12 a4508e8 0623ff0 3c35b12 0623ff0 3c35b12 a4508e8 0623ff0 a4508e8 3c35b12 0623ff0 3c35b12 a4508e8 0623ff0 3c35b12 0623ff0 3c35b12 a4508e8 0623ff0 a4508e8 3c35b12 a4508e8 0623ff0 a4508e8 0623ff0 3c35b12 0623ff0 3c35b12 0623ff0 3c35b12 0623ff0 3c35b12 0623ff0 3c35b12 a4508e8 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 | """Train and save the production XGBoost model on ALL ~1112 valid-target rows (incl. out-of-scope GPUs, mirroring what LOGO-CV folds saw), using notebooks/03_model_training.ipynb cell 3's exact feature encoding for GpuPredictor compatibility."""
from __future__ import annotations
import json
import logging
import stat as _stat
from pathlib import Path
import numpy as np
import pandas as pd
import xgboost as xgb
from src.data.manifest import corpus_sha256, load_manifest, sources_present, verify_manifest
from src.models.predictor import FEATURE_COLS
# Written by notebooks/03_model_training.ipynb (the only place that fits out-of-fold models); reading it here is optional/best-effort since train_and_save() trains the production model on ALL rows with no hold-out and so has no LOGO-CV metrics of its own.
LOGO_CV_METRICS_PATH = Path("data/models/logo_cv_metrics.json")
# Same cap predictor.py applies to feature_metadata.json — same trust tier (build artifact, not user input) and comparably small.
_MAX_VALIDATION_METRICS_BYTES: int = 1 * 1024 * 1024 # 1 MB
logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s")
log = logging.getLogger(__name__)
TARGET = "efficiency_ratio"
THROUGHPUT_COL = "throughput_tok_per_sec_per_gpu"
# Same hyperparameters as the training notebook's final model (no early stopping); n_estimators=250 is the midpoint of observed best_iteration across folds (77-341), conservative to avoid overfitting without an eval set.
PROD_PARAMS: dict = {
"n_estimators": 250,
"max_depth": 5,
"learning_rate": 0.03,
"subsample": 0.8,
"colsample_bytree": 0.8,
"min_child_weight": 5,
"tree_method": "hist",
"random_state": 42,
}
FEATURES_PATH = Path("data/processed/mlperf_features.parquet")
MODEL_DIR = Path("data/models")
def _encode(feat: pd.DataFrame) -> pd.DataFrame:
"""Apply the same categorical encoding as notebook cell 3."""
df = feat.copy()
df["scenario_offline"] = (df["scenario"] == "Offline").astype(int)
df["fw_tensorrt"] = (df["framework_family"] == "tensorrt").astype(int)
df["fw_vllm"] = (df["framework_family"] == "vllm").astype(int)
df["fw_rocm_other"] = (df["framework_family"] == "rocm_other").astype(int)
df["is_cdna4"] = (df["amd_arch_gen"] == 2).astype(int)
return df
def _load_validation_metrics(path: Path = LOGO_CV_METRICS_PATH) -> dict | None:
"""Best-effort load of the LOGO-CV metrics notebooks/03 writes: returns None (not a raised error) when absent (e.g. fresh clone, notebook never re-run), but once present gets the same symlink+size guard as every other integrity-sensitive file this codebase reads, so a corrupted or maliciously-symlinked file fails loudly rather than silently reading tampered content."""
try:
st = path.lstat()
except OSError:
return None
if _stat.S_ISLNK(st.st_mode):
raise ValueError(f"Validation metrics path is a symlink (refused): {path}")
if st.st_size > _MAX_VALIDATION_METRICS_BYTES:
raise ValueError(
f"Validation metrics file too large ({st.st_size} bytes > "
f"{_MAX_VALIDATION_METRICS_BYTES}): {path}"
)
with path.open() as f:
return json.load(f)
def train_and_save(
features_path: Path = FEATURES_PATH,
model_dir: Path = MODEL_DIR,
) -> xgb.XGBRegressor:
# Verify the corpus's source snapshots still match data_manifest.lock (warn-only, see manifest.py's docstring for why); result is captured so feature_metadata.json can record verification status, with n_sources_present/total disambiguating "clean" from "mostly unchecked" since absent gitignored sources are skipped not flagged; manifest loaded once and passed to both calls to avoid re-parsing the lock file (~1.2ms/call) per call site.
locked_manifest = load_manifest()
manifest_mismatches = verify_manifest(locked=locked_manifest)
n_sources_present, n_sources_total = sources_present(locked=locked_manifest)
# Hash the exact corpus file before reading it, so feature_metadata.json records the snapshot that actually produced the model.
corpus_hash = corpus_sha256(features_path)
feat = pd.read_parquet(features_path)
log.info("Loaded %d rows from %s", len(feat), features_path)
df = _encode(feat)
# Drop rows where target is NaN or infinite.
valid = df[TARGET].notna() & np.isfinite(df[TARGET])
df = df[valid].reset_index(drop=True)
log.info("Training rows (valid target): %d", len(df))
# Variance gate matching the training notebook's assertion; explicit raise (not assert) since assert is compiled away under python -O.
dead = [c for c in FEATURE_COLS if df[c].nunique() <= 1]
if dead:
raise ValueError(f"Dead features (constant across all training rows): {dead}")
X = df[FEATURE_COLS]
y = df[TARGET]
model = xgb.XGBRegressor(**PROD_PARAMS)
model.fit(X, y)
log.info("Model trained: %d trees, %d features", PROD_PARAMS["n_estimators"], len(FEATURE_COLS))
validation_metrics = _load_validation_metrics()
model_dir.mkdir(parents=True, exist_ok=True)
model_path = model_dir / "prophet_v1.json"
meta_path = model_dir / "feature_metadata.json"
# Write both artifacts atomically via temp-file + os.replace(), so a kill mid-write leaves the reader (GpuPredictor.__init__) seeing the old pair or the new pair, never a mismatched mix.
tmp_model = model_path.with_name(model_path.stem + ".tmp.json")
model.save_model(str(tmp_model))
tmp_model.replace(model_path)
log.info("Saved model → %s", model_path)
meta = {
"feature_cols": FEATURE_COLS,
"target": TARGET,
"model_version": "v1",
"prod_params": PROD_PARAMS,
"n_training_rows": len(df),
# Links this artifact to the exact corpus snapshot it trained on (lineage: prediction -> feature_metadata.json -> data_manifest.lock -> pinned MLPerf commits); manifest_verified confirms *this* corpus's pinning was actually checked clean, not just that a hash exists.
"corpus_sha256": corpus_hash,
"corpus_path": str(features_path),
"manifest_verified": len(manifest_mismatches) == 0,
# Disambiguates manifest_verified: "5/5" (every locked source present and clean) reads very differently from "1/5" (gitignored MLPerf mirrors simply absent, the normal fresh-clone/CI state), even though both give manifest_verified: true.
"manifest_sources_present": n_sources_present,
"manifest_sources_total": n_sources_total,
# GPUs with zero rows here extrapolate purely from specs; served via GpuPredictor.has_training_data() so callers can disclose this instead of presenting an extrapolated number with false confidence.
"trained_gpu_ids": sorted(df["canonical_gpu_id"].unique().tolist()),
# Per-GPU row counts behind GpuPredictor.training_data_tier(): a boolean "has any data" can't tell a GPU with 1 row from one with 1,000 — three of eight served GPUs (the AMD SKUs) sit under the 100-row-per-GPU floor despite having nonzero data.
"trained_gpu_row_counts": {
str(k): int(v) for k, v in df["canonical_gpu_id"].value_counts().items()
},
# LOGO-CV out-of-fold metrics from notebooks/03, not this run's own result (train_and_save() fits on ALL rows with no hold-out); None if the notebook wasn't re-run since the last corpus rebuild.
"validation_metrics": validation_metrics,
}
tmp_meta = meta_path.with_name(meta_path.name + ".tmp")
with tmp_meta.open("w") as f:
json.dump(meta, f, indent=2)
tmp_meta.replace(meta_path)
log.info("Saved metadata → %s", meta_path)
return model
if __name__ == "__main__":
train_and_save()
|