Trifecta-Lab / build_comparison_pdf.py
Brettapps's picture
Upload folder using huggingface_hub (part 20)
013874c verified
Raw History Blame Contribute Delete
9.07 kB
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""Build the trifecta comparison PDF for 2026-08-10 from _rerun_summary.json."""
from __future__ import annotations
import json
from datetime import datetime, timezone
from pathlib import Path
from reportlab.lib import colors
from reportlab.lib.pagesizes import A4
from reportlab.lib.styles import getSampleStyleSheet, ParagraphStyle
from reportlab.lib.units import mm
from reportlab.platypus import (
SimpleDocTemplate, Table, TableStyle, Paragraph, Spacer, HRFlowable,
)
BASE = Path(__file__).resolve().parent
S = json.loads((BASE / "data" / "predictions" / "_rerun_summary.json").read_text())
MODELS = ["Model A (default)", "Variant B (form-heavy)", "Variant C (place-heavy)"]
NEON = colors.HexColor("#16C47F")
DARK = colors.HexColor("#0F172A")
MUTED = colors.HexColor("#64748B")
LIGHT = colors.HexColor("#F1F5F9")
styles = getSampleStyleSheet()
H1 = ParagraphStyle("H1", parent=styles["Title"], fontSize=22, textColor=DARK, spaceAfter=2, leading=26)
SUB = ParagraphStyle("SUB", parent=styles["Normal"], fontSize=10, textColor=MUTED, spaceAfter=10)
H2 = ParagraphStyle("H2", parent=styles["Heading2"], fontSize=14, textColor=DARK, spaceBefore=10, spaceAfter=6)
BODY = ParagraphStyle("BODY", parent=styles["Normal"], fontSize=9.5, textColor=DARK, leading=13)
SMALL = ParagraphStyle("SMALL", parent=styles["Normal"], fontSize=8, textColor=MUTED, leading=10)
CELL = ParagraphStyle("CELL", parent=styles["Normal"], fontSize=8.5, textColor=DARK, leading=11)
CELLH = ParagraphStyle("CELLH", parent=styles["Normal"], fontSize=8.5, textColor=colors.white, leading=11)
def pct(a, b):
return f"{100*a/b:.0f}%" if b else "0%"
def build():
out = BASE / "data" / "pdfs" / "trifecta-comparison-2026-08-10.pdf"
doc = SimpleDocTemplate(str(out), pagesize=A4, title="Trifecta-Bro Comparison 2026-08-10",
author="Trifecta-Bro", leftMargin=16*mm, rightMargin=16*mm,
topMargin=16*mm, bottomMargin=14*mm)
story = []
story.append(Paragraph("Trifecta-Bro — Prediction vs Result Comparison", H1))
story.append(Paragraph(
f"Date 2026-08-10  |  Tracks: Dubbo, Kilcoy, Nowra  |  "
f"Generated {datetime.now(timezone.utc):%Y-%m-%d %H:%M UTC}  |  "
f"Engine: open-source TrifectaPredictor (LM Studio model v1 has no weights loaded)", SUB))
story.append(HRFlowable(width="100%", thickness=1.2, color=NEON, spaceAfter=8))
# Scorecard table
story.append(Paragraph("Scorecard — 20 scorable races (Dubbo R7 excluded: race time only)", H2))
head = ["Metric", "Model A (default)", "Variant B (form)", "Variant C (place)"]
rows = [head]
def r(metric, key, denom_key):
def g(m):
d = S["results"][m]
return f"{d[key]}/{d[denom_key]} ({pct(d[key], d[denom_key])})"
rows.append([metric, g(MODELS[0]), g(MODELS[1]), g(MODELS[2])])
def r1(metric, key):
rows.append([metric, str(S["results"][MODELS[0]][key]),
str(S["results"][MODELS[1]][key]), str(S["results"][MODELS[2]][key])])
r("Exact trifecta hit", "exact", "races")
r("Any box-of-3 (any order)", "box", "races")
r("Winner picked", "winner", "races")
r("Winner inside predicted top-3", "win_in_top3", "n_all")
r("Full 3 in predicted top-3", "full3_in_top3", "races3")
r("2-of-3 in predicted top-3", "two3_in_top3", "races3")
r1("Quinella hits (2-horse results)", "quinella")
t = Table(rows, colWidths=[58*mm, 38*mm, 38*mm, 38*mm])
t.setStyle(TableStyle([
("BACKGROUND", (0,0), (-1,0), DARK),
("TEXTCOLOR", (0,0), (-1,0), colors.white),
("FONTNAME", (0,0), (-1,0), "Helvetica-Bold"),
("FONTSIZE", (0,0), (-1,-1), 8.5),
("ROWBACKGROUNDS", (0,1), (-1,-1), [colors.white, LIGHT]),
("GRID", (0,0), (-1,-1), 0.4, colors.HexColor("#CBD5E1")),
("ALIGN", (1,0), (-1,-1), "CENTER"),
("VALIGN", (0,0), (-1,-1), "MIDDLE"),
("TOPPADDING", (0,0), (-1,-1), 3),
("BOTTOMPADDING", (0,0), (-1,-1), 3),
# highlight Model A column winner/exact rows
("BACKGROUND", (1,1), (1,3), colors.HexColor("#ECFDF5")),
]))
story.append(t)
story.append(Spacer(1, 6))
story.append(Paragraph(
"Model A (default weights) posted the best result: 4 winners and the lone exact trifecta. "
"Neither alternative weighting improved the strike rate — Variant B and C each picked 3 winners.", SMALL))
# Race-by-race
story.append(Paragraph("Race-by-Race — actual vs each model's primary pick", H2))
det = {m: {f"{d[0]}_R{d[1]}": d[3]["primary"] for d in S["details"][m] if d[3]} for m in MODELS}
actual = {
("Dubbo","1"):"1-7-9",("Dubbo","2"):"8-3-6",("Dubbo","3"):"5-1-4",("Dubbo","4"):"4-9-6",
("Dubbo","5"):"2-8",("Dubbo","6"):"13-11-9",("Dubbo","7"):"(time)",
("Kilcoy","1"):"7-8",("Kilcoy","2"):"6-9-4",("Kilcoy","3"):"2-4",("Kilcoy","4"):"6-1-4",
("Kilcoy","5"):"5-11-9",("Kilcoy","6"):"10-4",("Kilcoy","7"):"15-9-1",
("Nowra","1"):"6-8-1",("Nowra","2"):"9-11-1",("Nowra","3"):"3-5-1",("Nowra","4"):"4-11-2",
("Nowra","5"):"8-11-3",("Nowra","6"):"7-1-6",("Nowra","7"):"2-3-9",
}
keys = sorted(actual.keys(), key=lambda k: (k[0], int(k[1])))
head2 = ["Track", "R", "Actual", "A", "B", "C", "Hit"]
rdata = [head2]
for trk, rn in keys:
a = actual[(trk, rn)]
pa = det[MODELS[0]].get(f"{trk}_R{rn}", "--")
pb = det[MODELS[1]].get(f"{trk}_R{rn}", "--")
pc = det[MODELS[2]].get(f"{trk}_R{rn}", "--")
hit = ""
if a != "(time)":
av = [int(x) for x in a.split("-")]
for m, pk in ((MODELS[0],pa),(MODELS[1],pb),(MODELS[2],pc)):
if pk != "--":
pv = [int(x) for x in pk.split("-")]
if pv[0] == av[0]:
hit += "W" if m==MODELS[0] else ("B" if m==MODELS[1] else "C")
rdata.append([trk, rn, a, pa, pb, pc, hit or "-"])
t2 = Table(rdata, colWidths=[20*mm, 8*mm, 20*mm, 22*mm, 22*mm, 22*mm, 22*mm], repeatRows=1)
ts = [
("BACKGROUND", (0,0), (-1,0), DARK),
("TEXTCOLOR", (0,0), (-1,0), colors.white),
("FONTNAME", (0,0), (-1,0), "Helvetica-Bold"),
("FONTSIZE", (0,0), (-1,-1), 8),
("ROWBACKGROUNDS", (0,1), (-1,-1), [colors.white, LIGHT]),
("GRID", (0,0), (-1,-1), 0.3, colors.HexColor("#CBD5E1")),
("ALIGN", (1,0), (-1,-1), "CENTER"),
("TOPPADDING", (0,0), (-1,-1), 2.2),
("BOTTOMPADDING", (0,0), (-1,-1), 2.2),
]
# shade hit cells
for i, (trk, rn) in enumerate(keys, start=1):
a = actual[(trk, rn)]
if a == "(time)":
ts.append(("TEXTCOLOR", (6,i), (6,i), MUTED))
t2.setStyle(TableStyle(ts))
story.append(t2)
story.append(Spacer(1, 8))
# Read-through
story.append(Paragraph("Read-Through", H2))
for line in [
"1. Model A nailed Dubbo R1 outright (exact trifecta 1-7-9) and found 4 winners overall — "
"the strongest single-model run of the day.",
"2. Re-weighting the scorer toward recent form (Variant B) or toward place%+prize (Variant C) "
"did NOT lift the strike rate. The default weights already performed best.",
"3. The shared weakness is the SAME across all three: across the 16 complete 3-place results, "
"ZERO had all three actual trifecta runners sitting inside the predicted top-3, and only 4/16 "
"had even two of three. The models rarely select the correct SET of runners — they are "
"better at the front (35-45% of winners in the top-3) than at nailing the exotics.",
"4. This is a data/feature limitation of the open-source heuristic (no weights loaded for the "
"LM Studio model v1), not a tuning problem. Meaningful gains need either the trained "
"Brettapps/trifecta-bro/v1 weights, or richer inputs (speed ratings, sectional times, "
"jockey/trainer strike rates, track bias).",
"5. Caveats: several pasted results are 2 numbers only (Dubbo R5; Kilcoy R1/R3/R6) and were "
"judged as quinella/winner only. Dubbo R7 carried a race time with no finishing order.",
]:
story.append(Paragraph("• " + line, BODY))
story.append(Spacer(1, 3))
story.append(Spacer(1, 8))
story.append(HRFlowable(width="100%", thickness=0.6, color=MUTED, spaceAfter=4))
story.append(Paragraph(
"Method: predictions regenerated from the same FormFav runner data in predictions-2026-08-10.json. "
"Variant B = form x1.6, place x0.8, distance x1.2, barrier x0.3, prize x0.5. "
"Variant C = place x1.7, prize x1.3, track x1.2, condition x1.2, form x0.7, barrier x0.5. "
"Honest backtest — only the scoring weights differ.", SMALL))
doc.build(story)
print(f"wrote {out} ({out.stat().st_size} bytes)")
if __name__ == "__main__":
build()