worldcup-api / rule_based_model.py
wihaha's picture
First release to Hugging Face
a3d10df
Raw
History Blame Contribute Delete
3.43 kB
import numpy as np
class RuleBasedModel:
"""
A simple baseline model using Bradley-Terry curve for Elo differences.
"""
def __init__(self):
self.classes_ = np.array([0, 1, 2])
def fit(self, X, y):
pass # No training required for rule-based baseline
def predict_proba(self, X):
# Base Elo difference
# We assume home advantage is roughly +50 Elo
# Momentum diff is scaled and added to Elo diff
# Apply injury impact: injury_impact is in [-0.1, 0]. Assume -0.1 = -100 Elo points.
home_injury_penalty = X['injury_impact_home'] * 1000
away_injury_penalty = X['injury_impact_away'] * 1000
# Apply lineup strength: lineup_strength is in [0.8, 1.2].
# A 1.2 multiplier represents a much stronger squad than normal.
# Let's say +0.1 = +100 Elo points.
home_lineup_bonus = (X['lineup_strength_home'] - 1.0) * 1000
away_lineup_bonus = (X['lineup_strength_away'] - 1.0) * 1000
# Calculate modified Elo diff
base_elo_diff = X['elo_diff'] + (X['home_advantage'] * 50) + (X['momentum_diff'] * 100)
# Add home adjustments and subtract away adjustments
modified_elo_diff = base_elo_diff + home_injury_penalty + home_lineup_bonus - away_injury_penalty - away_lineup_bonus
# Expected win rate for home team (Bradley-Terry curve)
home_expected = 1.0 / (1.0 + 10.0 ** (-modified_elo_diff / 400.0))
# Empirical draw probability in football is roughly 25-30% on evenly matched teams,
# dropping off as teams become mismatched.
prob_draw = 0.28 * np.exp(-(modified_elo_diff ** 2) / (2 * 400**2))
# The remainder is split between home and away based on the expected score
remaining = 1.0 - prob_draw
prob_home = remaining * home_expected
prob_away = remaining * (1.0 - home_expected)
# Blend with odds implied probability if available
has_odds = (X['odds_implied_home_prob'] > 0) & (X['odds_implied_away_prob'] > 0)
# If odds exist, we do a 50/50 blend between our modified model and bookmaker odds
# Bookmakers don't explicitly give draw probability in the engineered features directly,
# but we can deduce it as 1 - odds_home - odds_away
odds_prob_home = X['odds_implied_home_prob']
odds_prob_away = X['odds_implied_away_prob']
odds_prob_draw = 1.0 - odds_prob_home - odds_prob_away
# Ensure we don't have negative probabilities due to floating point inaccuracies
odds_prob_draw = np.maximum(odds_prob_draw, 0.0)
final_prob_home = np.where(has_odds, 0.5 * prob_home + 0.5 * odds_prob_home, prob_home)
final_prob_away = np.where(has_odds, 0.5 * prob_away + 0.5 * odds_prob_away, prob_away)
final_prob_draw = np.where(has_odds, 0.5 * prob_draw + 0.5 * odds_prob_draw, prob_draw)
# Normalize just to be safe
total = final_prob_home + final_prob_away + final_prob_draw
final_prob_home /= total
final_prob_away /= total
final_prob_draw /= total
return np.column_stack([final_prob_away, final_prob_draw, final_prob_home])
def predict(self, X):
probs = self.predict_proba(X)
return np.argmax(probs, axis=1)