Trifecta-Lab / rerun_compare.py
Brettapps's picture
Upload folder using huggingface_hub (part 21)
e23172f verified
Raw History Blame Contribute Delete
11.3 kB
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""Re-run trifecta predictions with alternative scoring params and compare
against the pasted actual results for 2026-08-10.
Model A = the predictions already published in predictions-2026-08-10.json
(open-source TrifectaPredictor, default weights).
Variant B = form-heavy reweight (recent form boosted, barrier/prize damped).
Variant C = place-heavy reweight (place% + prize emphasised, barrier half).
All variants score the SAME runner data (from the prediction file's `form`
block) so the only variable is the scoring function. This is honest
backtesting, not cherry-picking.
"""
from __future__ import annotations
import json
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from src.prediction_model import Runner, Race, TrifectaPredictor
DATA = Path(__file__).resolve().parent / "data"
PRED_FILE = DATA / "predictions" / "predictions-2026-08-10.json"
# --- Actuals from the user paste (1st,2nd,3rd finishing order) ---------------
# Cells with a colon are race times -> no finishing order captured.
ACTUAL = {
("Dubbo", "1"): [1, 7, 9], ("Dubbo", "2"): [8, 3, 6], ("Dubbo", "3"): [5, 1, 4],
("Dubbo", "4"): [4, 9, 6], ("Dubbo", "5"): [2, 8], ("Dubbo", "6"): [13, 11, 9],
("Dubbo", "7"): None,
("Kilcoy", "1"): [7, 8], ("Kilcoy", "2"): [6, 9, 4], ("Kilcoy", "3"): [2, 4],
("Kilcoy", "4"): [6, 1, 4], ("Kilcoy", "5"): [5, 11, 9], ("Kilcoy", "6"): [10, 4],
("Kilcoy", "7"): [15, 9, 1],
("Nowra", "1"): [6, 8, 1], ("Nowra", "2"): [9, 11, 1], ("Nowra", "3"): [3, 5, 1],
("Nowra", "4"): [4, 11, 2], ("Nowra", "5"): [8, 11, 3], ("Nowra", "6"): [7, 1, 6],
("Nowra", "7"): [2, 3, 9],
}
# --- Alternative scoring variants -------------------------------------------
class VariantPredictor(TrifectaPredictor):
"""TrifectaPredictor with configurable component weights."""
def __init__(self, w=None):
self.w = {
"form": 1.0, "win_pct": 1.0, "place_pct": 1.0, "track": 1.0,
"distance": 1.0, "condition": 1.0, "barrier": 1.0, "prize": 1.0,
}
if w:
self.w.update(w)
def _score_runner(self, runner):
w = self.w
score = 0.0
form = str(runner.form or runner.last20Starts or "")
recent = form[-5:] if len(form) > 5 else form
score += min((recent.count("1") * 8 + recent.count("2") * 4 + recent.count("3") * 4), 25) * w["form"]
overall = runner.stats.get("overall", {})
starts = overall.get("starts", 0) or 0
win_pct = overall.get("winPercent", 0) or 0
place_pct = overall.get("placePercent", 0) or 0
score += win_pct * 20 * w["win_pct"]
score += place_pct * 10 * w["place_pct"]
track_stats = runner.stats.get("track", {})
track_starts = track_stats.get("starts", 0) or 0
track_places = track_stats.get("places", 0) or 0
score += min((track_places / max(track_starts, 1)) * 10, 10) * w["track"]
dist_stats = runner.stats.get("distance", {})
dist_starts = dist_stats.get("starts", 0) or 0
dist_places = dist_stats.get("places", 0) or 0
score += min((dist_places / max(dist_starts, 1)) * 8, 8) * w["distance"]
cond_stats = runner.stats.get("conditions", {})
for data in cond_stats.values():
c_starts = data.get("starts", 0) or 0
c_places = data.get("places", 0) or 0
score += min((c_places / max(c_starts, 1)) * 8, 8) * w["condition"]
try:
barrier = int(runner.barrier) if runner.barrier else 5
score += max(0, 5 - abs(barrier - 5)) * w["barrier"]
except Exception:
score += 3 * w["barrier"]
try:
prize = float(str(runner.careerPrizeMoney).replace("$", "").replace(",", ""))
score += min(prize / 20000, 5) * w["prize"]
except Exception:
pass
return min(round(score, 1), 100)
VARIANTS = {
"Variant B (form-heavy)": VariantPredictor({
"form": 1.6, "win_pct": 1.0, "place_pct": 0.8,
"track": 1.0, "distance": 1.2, "condition": 1.0,
"barrier": 0.3, "prize": 0.5,
}),
"Variant C (place-heavy)": VariantPredictor({
"form": 0.7, "win_pct": 1.0, "place_pct": 1.7,
"track": 1.2, "distance": 1.0, "condition": 1.2,
"barrier": 0.5, "prize": 1.3,
}),
}
def parse(s):
return [int(x) for x in s.split("-")]
def build_race(form) -> Race:
runners = []
for r in form.get("runners", []):
runners.append(Runner(
number=r.get("number"), name=r.get("name", ""),
jockey=r.get("jockey", ""), trainer=r.get("trainer", ""),
weight=r.get("weight"), barrier=r.get("barrier"),
form=r.get("form", ""), last20Starts=r.get("last20Starts", ""),
careerPrizeMoney=r.get("careerPrizeMoney", "$0"),
stats=r.get("stats", {}),
))
return Race(
date=form.get("date"), track=form.get("track"), track_slug=form.get("track_slug", ""),
race_number=str(form.get("raceNumber")), race_name=form.get("raceName", ""),
distance=form.get("distance", ""), condition=form.get("condition", ""),
weather=form.get("weather", ""), race_class=form.get("raceClass", ""),
start_time=form.get("startTime", ""), prize_money=form.get("prizeMoney", ""),
number_of_runners=form.get("numberOfRunners", 0), runners=runners,
)
def score_predictions(pred_map) -> dict:
"""pred_map: (track,rn) -> {'primary','secondary','value','top3'(list of nums)}"""
tot_exact = tot_box = tot_win = tot_q = races = 0
full3 = two3 = win_in_top3 = races3 = 0
details = []
for (track, rn), a in ACTUAL.items():
p = pred_map.get((track, rn))
if not p:
continue
prim = parse(p["primary"]); sec = parse(p["secondary"] or ""); val = parse(p["value"] or "")
top3 = p["top3"]
if a is None:
details.append((track, rn, None, p, "time-only"))
continue
races += 1
exact = any(t == a[:3] for t in [prim, sec, val])
box = any(set(t[:3]) == set(a[:3]) for t in [prim, sec, val] if len(t) >= 3)
win = a[0] in {prim[0], sec[0], val[0]}
q = (len(a) == 2 and any(sorted(t[:2]) == sorted(a) for t in [prim, sec, val]))
tot_exact += exact; tot_box += box; tot_win += win; tot_q += q
if len(a) == 3:
races3 += 1
hits = len(set(a) & set(top3))
if hits == 3:
full3 += 1
elif hits == 2:
two3 += 1
if a[0] in top3:
win_in_top3 += 1
tag = []
if exact: tag.append("EXACT")
if box: tag.append("BOX")
if win: tag.append("WIN")
if q: tag.append("QUIN")
details.append((track, rn, a, p, ",".join(tag) if tag else "miss"))
summary = {
"races": races, "exact": tot_exact, "box": tot_box,
"winner": tot_win, "quinella": tot_q,
"races3": races3, "full3_in_top3": full3, "two3_in_top3": two3,
"win_in_top3": win_in_top3, "n_all": sum(1 for v in ACTUAL.values() if v),
"details": details,
}
return summary
def main():
data = json.loads(PRED_FILE.read_text())
# Model A: read stored predictions verbatim
pred_a = {}
for r in data["races"]:
p = r["prediction"]
pred_a[(r["track"], str(r["race_number"]))] = {
"primary": p["primary"], "secondary": p.get("secondary"),
"value": p.get("value"), "top3": [t["number"] for t in p["top3"]],
}
summary_a = score_predictions(pred_a)
# Build races once
races = [build_race(r["form"]) for r in data["races"]]
results = {"Model A (default)": summary_a}
out_variants = {}
for name, predictor in VARIANTS.items():
pred_map = {}
for rc in races:
key = (rc.track, rc.race_number)
if key not in ACTUAL or ACTUAL[key] is None:
continue
pr = predictor.predict(rc)
if "error" in pr:
continue
pred_map[key] = {
"primary": pr["primary"], "secondary": pr.get("secondary"),
"value": pr.get("value"), "top3": [t["number"] for t in pr["top3"]],
}
summary = score_predictions(pred_map)
results[name] = summary
# Save variant predictions for transparency
out_variants[name] = {f"{k[0]}_R{k[1]}": v for k, v in pred_map.items()}
# Persist variant predictions
for name, mp in out_variants.items():
slug = name.split("(")[0].strip().replace(" ", "_").lower()
out = DATA / "predictions" / f"predictions-2026-08-10-{slug}.json"
out.write_text(json.dumps(mp, indent=2))
print(f"wrote {out}")
# Print scorecards
print("\n" + "=" * 78)
print("SCORECARD COMPARISON (2026-08-10, 20 scorable races)")
print("=" * 78)
for name, s in results.items():
print(f"\n{name}")
print(f" exact trifecta ... {s['exact']}/{s['races']} ({100*s['exact']/s['races']:.0f}%)")
print(f" any box-of-3 .... {s['box']}/{s['races']} ({100*s['box']/s['races']:.0f}%)")
print(f" winner picked ... {s['winner']}/{s['races']} ({100*s['winner']/s['races']:.0f}%)")
print(f" quinella (2-no) . {s['quinella']}")
print(f" winner in top3 .. {s['win_in_top3']}/{s['n_all']} ({100*s['win_in_top3']/s['n_all']:.0f}%)")
if s['races3']:
print(f" full 3 in top3 ... {s['full3_in_top3']}/{s['races3']} | 2-of-3 in top3 {s['two3_in_top3']}/{s['races3']}")
# Race-by-race comparison table
print("\n" + "=" * 78)
print("RACE-BY-RACE (actual | A | B | C)")
print("=" * 78)
keys = sorted(ACTUAL.keys(), key=lambda k: (k[0], int(k[1])))
smap = {n: results[n] for n in results}
for key in keys:
track, rn = key
a = ACTUAL[key]
astr = "-".join(map(str, a)) if a else "(time)"
row = []
for n in ["Model A (default)", "Variant B (form-heavy)", "Variant C (place-heavy)"]:
det = next((d for d in smap[n]["details"] if d[0] == track and d[1] == rn), None)
if det and det[3]:
row.append(det[3]["primary"])
else:
row.append("--")
verdict = ""
# mark which models hit winner
for n in ["Model A (default)", "Variant B (form-heavy)", "Variant C (place-heavy)"]:
det = next((d for d in smap[n]["details"] if d[0] == track and d[1] == rn), None)
if det and det[4] not in (None, "time-only", "miss"):
verdict += f" {n.split(' ')[1]}:{det[4]}"
print(f" {track:7s} R{rn} {astr:8s} | {row[0]:7s} {row[1]:7s} {row[2]:7s}{verdict}")
# Save machine-readable results for the PDF builder
(DATA / "predictions" / "_rerun_summary.json").write_text(json.dumps({
"date": "2026-08-10",
"results": {n: {k: v for k, v in s.items() if k != "details"} for n, s in results.items()},
"details": {n: s["details"] for n, s in results.items()},
}, indent=2, default=str))
print("\nwrote data/predictions/_rerun_summary.json")
if __name__ == "__main__":
main()