#!/usr/bin/env python3 # -*- coding: utf-8 -*- """Re-run trifecta predictions with alternative scoring params and compare against the pasted actual results for 2026-08-10. Model A = the predictions already published in predictions-2026-08-10.json (open-source TrifectaPredictor, default weights). Variant B = form-heavy reweight (recent form boosted, barrier/prize damped). Variant C = place-heavy reweight (place% + prize emphasised, barrier half). All variants score the SAME runner data (from the prediction file's `form` block) so the only variable is the scoring function. This is honest backtesting, not cherry-picking. """ from __future__ import annotations import json import sys from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent)) from src.prediction_model import Runner, Race, TrifectaPredictor DATA = Path(__file__).resolve().parent / "data" PRED_FILE = DATA / "predictions" / "predictions-2026-08-10.json" # --- Actuals from the user paste (1st,2nd,3rd finishing order) --------------- # Cells with a colon are race times -> no finishing order captured. ACTUAL = { ("Dubbo", "1"): [1, 7, 9], ("Dubbo", "2"): [8, 3, 6], ("Dubbo", "3"): [5, 1, 4], ("Dubbo", "4"): [4, 9, 6], ("Dubbo", "5"): [2, 8], ("Dubbo", "6"): [13, 11, 9], ("Dubbo", "7"): None, ("Kilcoy", "1"): [7, 8], ("Kilcoy", "2"): [6, 9, 4], ("Kilcoy", "3"): [2, 4], ("Kilcoy", "4"): [6, 1, 4], ("Kilcoy", "5"): [5, 11, 9], ("Kilcoy", "6"): [10, 4], ("Kilcoy", "7"): [15, 9, 1], ("Nowra", "1"): [6, 8, 1], ("Nowra", "2"): [9, 11, 1], ("Nowra", "3"): [3, 5, 1], ("Nowra", "4"): [4, 11, 2], ("Nowra", "5"): [8, 11, 3], ("Nowra", "6"): [7, 1, 6], ("Nowra", "7"): [2, 3, 9], } # --- Alternative scoring variants ------------------------------------------- class VariantPredictor(TrifectaPredictor): """TrifectaPredictor with configurable component weights.""" def __init__(self, w=None): self.w = { "form": 1.0, "win_pct": 1.0, "place_pct": 1.0, "track": 1.0, "distance": 1.0, "condition": 1.0, "barrier": 1.0, "prize": 1.0, } if w: self.w.update(w) def _score_runner(self, runner): w = self.w score = 0.0 form = str(runner.form or runner.last20Starts or "") recent = form[-5:] if len(form) > 5 else form score += min((recent.count("1") * 8 + recent.count("2") * 4 + recent.count("3") * 4), 25) * w["form"] overall = runner.stats.get("overall", {}) starts = overall.get("starts", 0) or 0 win_pct = overall.get("winPercent", 0) or 0 place_pct = overall.get("placePercent", 0) or 0 score += win_pct * 20 * w["win_pct"] score += place_pct * 10 * w["place_pct"] track_stats = runner.stats.get("track", {}) track_starts = track_stats.get("starts", 0) or 0 track_places = track_stats.get("places", 0) or 0 score += min((track_places / max(track_starts, 1)) * 10, 10) * w["track"] dist_stats = runner.stats.get("distance", {}) dist_starts = dist_stats.get("starts", 0) or 0 dist_places = dist_stats.get("places", 0) or 0 score += min((dist_places / max(dist_starts, 1)) * 8, 8) * w["distance"] cond_stats = runner.stats.get("conditions", {}) for data in cond_stats.values(): c_starts = data.get("starts", 0) or 0 c_places = data.get("places", 0) or 0 score += min((c_places / max(c_starts, 1)) * 8, 8) * w["condition"] try: barrier = int(runner.barrier) if runner.barrier else 5 score += max(0, 5 - abs(barrier - 5)) * w["barrier"] except Exception: score += 3 * w["barrier"] try: prize = float(str(runner.careerPrizeMoney).replace("$", "").replace(",", "")) score += min(prize / 20000, 5) * w["prize"] except Exception: pass return min(round(score, 1), 100) VARIANTS = { "Variant B (form-heavy)": VariantPredictor({ "form": 1.6, "win_pct": 1.0, "place_pct": 0.8, "track": 1.0, "distance": 1.2, "condition": 1.0, "barrier": 0.3, "prize": 0.5, }), "Variant C (place-heavy)": VariantPredictor({ "form": 0.7, "win_pct": 1.0, "place_pct": 1.7, "track": 1.2, "distance": 1.0, "condition": 1.2, "barrier": 0.5, "prize": 1.3, }), } def parse(s): return [int(x) for x in s.split("-")] def build_race(form) -> Race: runners = [] for r in form.get("runners", []): runners.append(Runner( number=r.get("number"), name=r.get("name", ""), jockey=r.get("jockey", ""), trainer=r.get("trainer", ""), weight=r.get("weight"), barrier=r.get("barrier"), form=r.get("form", ""), last20Starts=r.get("last20Starts", ""), careerPrizeMoney=r.get("careerPrizeMoney", "$0"), stats=r.get("stats", {}), )) return Race( date=form.get("date"), track=form.get("track"), track_slug=form.get("track_slug", ""), race_number=str(form.get("raceNumber")), race_name=form.get("raceName", ""), distance=form.get("distance", ""), condition=form.get("condition", ""), weather=form.get("weather", ""), race_class=form.get("raceClass", ""), start_time=form.get("startTime", ""), prize_money=form.get("prizeMoney", ""), number_of_runners=form.get("numberOfRunners", 0), runners=runners, ) def score_predictions(pred_map) -> dict: """pred_map: (track,rn) -> {'primary','secondary','value','top3'(list of nums)}""" tot_exact = tot_box = tot_win = tot_q = races = 0 full3 = two3 = win_in_top3 = races3 = 0 details = [] for (track, rn), a in ACTUAL.items(): p = pred_map.get((track, rn)) if not p: continue prim = parse(p["primary"]); sec = parse(p["secondary"] or ""); val = parse(p["value"] or "") top3 = p["top3"] if a is None: details.append((track, rn, None, p, "time-only")) continue races += 1 exact = any(t == a[:3] for t in [prim, sec, val]) box = any(set(t[:3]) == set(a[:3]) for t in [prim, sec, val] if len(t) >= 3) win = a[0] in {prim[0], sec[0], val[0]} q = (len(a) == 2 and any(sorted(t[:2]) == sorted(a) for t in [prim, sec, val])) tot_exact += exact; tot_box += box; tot_win += win; tot_q += q if len(a) == 3: races3 += 1 hits = len(set(a) & set(top3)) if hits == 3: full3 += 1 elif hits == 2: two3 += 1 if a[0] in top3: win_in_top3 += 1 tag = [] if exact: tag.append("EXACT") if box: tag.append("BOX") if win: tag.append("WIN") if q: tag.append("QUIN") details.append((track, rn, a, p, ",".join(tag) if tag else "miss")) summary = { "races": races, "exact": tot_exact, "box": tot_box, "winner": tot_win, "quinella": tot_q, "races3": races3, "full3_in_top3": full3, "two3_in_top3": two3, "win_in_top3": win_in_top3, "n_all": sum(1 for v in ACTUAL.values() if v), "details": details, } return summary def main(): data = json.loads(PRED_FILE.read_text()) # Model A: read stored predictions verbatim pred_a = {} for r in data["races"]: p = r["prediction"] pred_a[(r["track"], str(r["race_number"]))] = { "primary": p["primary"], "secondary": p.get("secondary"), "value": p.get("value"), "top3": [t["number"] for t in p["top3"]], } summary_a = score_predictions(pred_a) # Build races once races = [build_race(r["form"]) for r in data["races"]] results = {"Model A (default)": summary_a} out_variants = {} for name, predictor in VARIANTS.items(): pred_map = {} for rc in races: key = (rc.track, rc.race_number) if key not in ACTUAL or ACTUAL[key] is None: continue pr = predictor.predict(rc) if "error" in pr: continue pred_map[key] = { "primary": pr["primary"], "secondary": pr.get("secondary"), "value": pr.get("value"), "top3": [t["number"] for t in pr["top3"]], } summary = score_predictions(pred_map) results[name] = summary # Save variant predictions for transparency out_variants[name] = {f"{k[0]}_R{k[1]}": v for k, v in pred_map.items()} # Persist variant predictions for name, mp in out_variants.items(): slug = name.split("(")[0].strip().replace(" ", "_").lower() out = DATA / "predictions" / f"predictions-2026-08-10-{slug}.json" out.write_text(json.dumps(mp, indent=2)) print(f"wrote {out}") # Print scorecards print("\n" + "=" * 78) print("SCORECARD COMPARISON (2026-08-10, 20 scorable races)") print("=" * 78) for name, s in results.items(): print(f"\n{name}") print(f" exact trifecta ... {s['exact']}/{s['races']} ({100*s['exact']/s['races']:.0f}%)") print(f" any box-of-3 .... {s['box']}/{s['races']} ({100*s['box']/s['races']:.0f}%)") print(f" winner picked ... {s['winner']}/{s['races']} ({100*s['winner']/s['races']:.0f}%)") print(f" quinella (2-no) . {s['quinella']}") print(f" winner in top3 .. {s['win_in_top3']}/{s['n_all']} ({100*s['win_in_top3']/s['n_all']:.0f}%)") if s['races3']: print(f" full 3 in top3 ... {s['full3_in_top3']}/{s['races3']} | 2-of-3 in top3 {s['two3_in_top3']}/{s['races3']}") # Race-by-race comparison table print("\n" + "=" * 78) print("RACE-BY-RACE (actual | A | B | C)") print("=" * 78) keys = sorted(ACTUAL.keys(), key=lambda k: (k[0], int(k[1]))) smap = {n: results[n] for n in results} for key in keys: track, rn = key a = ACTUAL[key] astr = "-".join(map(str, a)) if a else "(time)" row = [] for n in ["Model A (default)", "Variant B (form-heavy)", "Variant C (place-heavy)"]: det = next((d for d in smap[n]["details"] if d[0] == track and d[1] == rn), None) if det and det[3]: row.append(det[3]["primary"]) else: row.append("--") verdict = "" # mark which models hit winner for n in ["Model A (default)", "Variant B (form-heavy)", "Variant C (place-heavy)"]: det = next((d for d in smap[n]["details"] if d[0] == track and d[1] == rn), None) if det and det[4] not in (None, "time-only", "miss"): verdict += f" {n.split(' ')[1]}:{det[4]}" print(f" {track:7s} R{rn} {astr:8s} | {row[0]:7s} {row[1]:7s} {row[2]:7s}{verdict}") # Save machine-readable results for the PDF builder (DATA / "predictions" / "_rerun_summary.json").write_text(json.dumps({ "date": "2026-08-10", "results": {n: {k: v for k, v in s.items() if k != "details"} for n, s in results.items()}, "details": {n: s["details"] for n, s in results.items()}, }, indent=2, default=str)) print("\nwrote data/predictions/_rerun_summary.json") if __name__ == "__main__": main()