"""Metadata para la web (ligas, temporadas, equipos, partidos). Dos fuentes, ambas bundleadas: - ``matches_catalog.parquet`` (de los nombres de los xlsx en Azure): cubre TODAS las ligas Opta (liga/temporada/fecha/local/visitante). Alimenta los selectores. - ``modeling/attack_prediction_dataset.parquet`` (dataset del modelo, 9 ligas): aporta el ``matchId`` real, necesario para el post-partido y la predicción. La GENERACIÓN de reportes sigue necesitando el preprocessed (Azure); por eso ``routes_meta`` interseca estas ligas con lo que realmente está subido. """ from __future__ import annotations import threading from datetime import date import pandas as pd from racing_reports import vendor_env _LOCK = threading.Lock() _CACHE: dict[str, object] = {} _CATALOG_PATH = vendor_env.DATA_DIR / "matches_catalog.parquet" _DATASET_PATH = vendor_env.DATA_DIR / "modeling" / "attack_prediction_dataset.parquet" _DS_COLS = ["matchId", "league", "season", "fecha", "team_name", "opponent_name", "is_home", "teamId", "opponent_team_id"] def _catalog() -> pd.DataFrame: with _LOCK: df = _CACHE.get("catalog") if df is None: if _CATALOG_PATH.exists(): df = pd.read_parquet(_CATALOG_PATH) df["league"] = df["league"].astype(str) df["season"] = df["season"].astype(str) else: df = pd.DataFrame(columns=["league", "season", "date", "home_team", "away_team"]) _CACHE["catalog"] = df return df def _dataset() -> pd.DataFrame: with _LOCK: df = _CACHE.get("dataset") if df is None: if _DATASET_PATH.exists(): cols = [c for c in _DS_COLS if c] df = pd.read_parquet(_DATASET_PATH, columns=cols) df["fecha"] = pd.to_datetime(df["fecha"], errors="coerce") for c in ("league", "season"): df[c] = df[c].astype(str) else: df = pd.DataFrame(columns=_DS_COLS) _CACHE["dataset"] = df return df def _model_leagues() -> set: d = _dataset() return set(d[d["team_name"].notna()]["league"].unique()) if len(d) else set() def has_dataset() -> bool: return _CATALOG_PATH.exists() or _DATASET_PATH.exists() def max_date() -> str: c = _catalog() if len(c): return str(c["date"].max()) d = _dataset() m = d["fecha"].max() if len(d) else None return m.strftime("%Y-%m-%d") if (m is not None and pd.notna(m)) else "" def league_seasons() -> dict[str, list[str]]: """{liga: [temporadas]} usables. Del catálogo (todas las ligas) ∪ dataset con nombres.""" out: dict[str, set] = {} c = _catalog() for league, season in zip(c["league"], c["season"]): out.setdefault(str(league), set()).add(str(season)) d = _dataset() named = d[d["team_name"].notna()] for league, season in zip(named["league"], named["season"]): out.setdefault(str(league), set()).add(str(season)) return {lg: sorted(ss, reverse=True) for lg, ss in sorted(out.items())} def team_ids(league: str, season: str) -> dict[str, str]: """Mapa nombre de equipo → teamId para una liga-temporada, desde el dataset del modelo (trae teamId + opponent_team_id). {} si la liga no está en el dataset.""" d = _dataset() if d.empty or "teamId" not in d.columns: return {} sub = d[(d["league"] == league) & (d["season"] == str(season))] out: dict[str, str] = {} for nm, tid in zip(sub["team_name"], sub.get("teamId")): if pd.notna(nm) and pd.notna(tid): out.setdefault(str(nm), str(tid)) if "opponent_team_id" in sub.columns: for nm, tid in zip(sub["opponent_name"], sub["opponent_team_id"]): if pd.notna(nm) and pd.notna(tid): out.setdefault(str(nm), str(tid)) return out def teams(league: str, season: str) -> list[str]: c = _catalog() sub = c[(c["league"] == league) & (c["season"] == str(season))] if len(sub): names = set(sub["home_team"].dropna()) | set(sub["away_team"].dropna()) return sorted(str(n) for n in names) # fallback al dataset d = _dataset() sd = d[(d["league"] == league) & (d["season"] == str(season))] names = set(sd["team_name"].dropna()) | set(sd["opponent_name"].dropna()) return sorted(str(n) for n in names) def _dataset_matches(league: str, season: str) -> list[dict]: d = _dataset() sub = d[(d["league"] == league) & (d["season"] == str(season)) & (d["is_home"] == True)] # noqa: E712 sub = sub.drop_duplicates("matchId") today = pd.Timestamp(date.today()) rows = [] for _, r in sub.sort_values("fecha", ascending=False).iterrows(): dt = r["fecha"] played = pd.notna(dt) and dt <= today rows.append({"match_id": str(r["matchId"]), "date": dt.strftime("%Y-%m-%d") if pd.notna(dt) else "", "home_team": str(r["team_name"]), "away_team": str(r["opponent_name"]), "status": "played" if played else "future"}) return rows def matches(league: str, season: str, played_only: bool = False) -> list[dict]: """Lista de partidos. Para ligas del modelo, con matchId real (post-partido); para el resto, del catálogo (sin matchId).""" today = date.today().isoformat() c = _catalog() sub = c[(c["league"] == league) & (c["season"] == str(season))].drop_duplicates(["date", "home_team", "away_team"]) cat_rows = [{"match_id": "", "date": str(r["date"])[:10], "home_team": str(r["home_team"]), "away_team": str(r["away_team"]), "status": "played" if str(r["date"])[:10] <= today else "future"} for _, r in sub.iterrows()] if league in _model_leagues(): # Dataset del modelo: trae matchId, pero puede estar cortado. Se completa con # el catálogo (cubre hasta la última fecha) — esos van sin matchId y el runner # lo resuelve del preprocessed al generar. rows = _dataset_matches(league, season) have = {(m["date"], m["home_team"], m["away_team"]) for m in rows} rows += [m for m in cat_rows if (m["date"], m["home_team"], m["away_team"]) not in have] rows.sort(key=lambda m: m["date"], reverse=True) else: rows = sorted(cat_rows, key=lambda m: m["date"], reverse=True) if played_only: rows = [m for m in rows if m["status"] == "played"] return rows def find_match(league: str, season: str, match_id: str) -> dict: for m in matches(league, season): if m["match_id"] and m["match_id"] == str(match_id): return m raise ValueError(f"No se encontró el partido {match_id} en {league} {season}.") def resolve_pair(league: str, season: str, team_a: str, team_b: str) -> list[dict]: a, b = team_a.strip().casefold(), team_b.strip().casefold() return [m for m in matches(league, season) if {m["home_team"].casefold(), m["away_team"].casefold()} == {a, b}]