numerai-weekly-v4 / inference.py
Noptus's picture
Add reproducible v4 weekly champion release
e9fa286 verified
Raw
History Blame Contribute Delete
5.38 kB
#!/usr/bin/env python3
"""Standalone, credential-free inference for the Numerai weekly v4 bundle."""
from __future__ import annotations
import argparse
from pathlib import Path
from typing import Any
import joblib
import numpy as np
import pandas as pd
from scipy.stats import norm, rankdata
REQUIRED_COMPONENTS = (
"benchmark_era_boost",
"multi_target",
"residual",
"catboost",
)
def load_bundle(path: str | Path) -> dict[str, Any]:
"""Load a trusted model bundle and validate its public inference contract."""
bundle = joblib.load(Path(path))
if not isinstance(bundle, dict):
raise TypeError("expected a dictionary model bundle")
missing = [name for name in REQUIRED_COMPONENTS if name not in bundle]
if missing:
raise ValueError(f"bundle is missing components: {', '.join(missing)}")
if "calibrated_weights" not in bundle or "config" not in bundle:
raise ValueError("bundle is missing calibrated_weights or config")
return bundle
def _columns(bundle: dict[str, Any]) -> tuple[list[str], list[str], list[str]]:
config = bundle["config"]
features = config.get("features")
benchmark_columns = config.get("bench_cols")
all_columns = config.get("all_feature_cols")
if not all(isinstance(value, list) for value in (features, benchmark_columns, all_columns)):
raise ValueError("bundle config does not contain serialized feature-name lists")
return features, benchmark_columns, all_columns
def _gaussianize(values: np.ndarray) -> np.ndarray:
ranked = rankdata(values, method="average") / (len(values) + 1)
return norm.ppf(ranked)
def predict(frame: pd.DataFrame, bundle: dict[str, Any]) -> np.ndarray:
"""Return the exact pre-neutralization Gaussian v4 ensemble prediction."""
features, _, all_columns = _columns(bundle)
missing = sorted(set(all_columns) - set(frame.columns))
if missing:
preview = ", ".join(missing[:8])
raise ValueError(f"input is missing {len(missing)} columns; first missing: {preview}")
n_rows = len(frame)
if n_rows < 2:
raise ValueError("at least two rows are required for cross-sectional ranking")
x_full = frame[all_columns].to_numpy()
x_features = frame[features].to_numpy()
components: dict[str, np.ndarray] = {}
models = bundle["benchmark_era_boost"]
components["benchmark_era_boost"] = np.mean(
[model.predict(x_full) for model in models], axis=0
)
target_predictions = [
rankdata(model.predict(x_full), method="average") / n_rows
for model in bundle["multi_target"].values()
]
components["multi_target"] = np.mean(target_predictions, axis=0)
models = bundle["residual"]
components["residual"] = np.mean(
[model.predict(x_features) for model in models], axis=0
)
models = bundle["catboost"]
components["catboost"] = np.mean(
[model.predict(x_full) for model in models], axis=0
)
for optional_name in ("xgboost", "lgb_dart"):
models = bundle.get(optional_name)
if models:
components[optional_name] = np.mean(
[model.predict(x_full) for model in models], axis=0
)
horizon_models = bundle.get("horizon60")
if horizon_models:
components["horizon60"] = np.mean(
[
rankdata(model.predict(x_full), method="average") / n_rows
for model in horizon_models.values()
],
axis=0,
)
weights = {
name: float(weight)
for name, weight in bundle["calibrated_weights"].items()
if name in components
}
total_weight = sum(weights.values())
if total_weight <= 0:
raise ValueError("bundle has no positively weighted active components")
ensemble = np.zeros(n_rows, dtype=np.float64)
for name, weight in weights.items():
component_rank = rankdata(components[name], method="average") / (n_rows + 1)
ensemble += (weight / total_weight) * component_rank
return _gaussianize(ensemble)
def predict_ranked(frame: pd.DataFrame, bundle: dict[str, Any]) -> np.ndarray:
"""Return submission-shaped predictions strictly between zero and one."""
gaussian_prediction = predict(frame, bundle)
return rankdata(gaussian_prediction, method="average") / (len(frame) + 1)
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--model", type=Path, required=True)
parser.add_argument("--live", type=Path, required=True)
parser.add_argument("--benchmarks", type=Path, required=True)
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
bundle = load_bundle(args.model)
live = pd.read_parquet(args.live)
benchmarks = pd.read_parquet(args.benchmarks)
benchmark_columns = [column for column in benchmarks.columns if column != "era"]
live = live.join(benchmarks[benchmark_columns], how="left")
predictions = predict_ranked(live, bundle)
output = pd.DataFrame({"prediction": predictions}, index=live.index)
output.index.name = "id"
args.output.parent.mkdir(parents=True, exist_ok=True)
output.to_csv(args.output)
print(f"wrote {len(output):,} predictions to {args.output}")
return 0
if __name__ == "__main__":
raise SystemExit(main())