Spaces:
Running
Running
| """Precalcula, por equipo (liga, temporada), su vector de features rolling 'a la fecha' | |
| (la última fila disponible) para alimentar la predicción de cruce A-vs-B en el reporte | |
| pre-partido, sin tener que re-correr todo el pipeline en runtime. | |
| Sale: | |
| vendor/data/modeling/team_features_latest.parquet (1 fila por equipo-temporada) | |
| vendor/data/modeling/matchup_feat_cols.json (orden exacto de features del modelo) | |
| Uso: python scripts/build_matchup_features.py | |
| """ | |
| from __future__ import annotations | |
| import importlib.util | |
| import json | |
| from pathlib import Path | |
| import pandas as pd | |
| ROOT = Path(__file__).resolve().parents[1] | |
| spec = importlib.util.spec_from_file_location("tm", ROOT / "scripts" / "train_models.py") | |
| tm = importlib.util.module_from_spec(spec); spec.loader.exec_module(tm) | |
| OUT = ROOT / "vendor" / "data" / "modeling" | |
| def main() -> None: | |
| df = pd.read_parquet(tm.DATA) | |
| blk = df.assign(_b=df[tm.DEF].sum(axis=1) > 0).groupby("Competencia")["_b"].mean() | |
| df = df[df["Competencia"].isin(blk[blk > 0.5].index)].copy() | |
| m, feat = tm._assemble(df) | |
| m = m.merge(df[["matchId", "teamId", "fecha", "TeamName"]].drop_duplicates(), | |
| on=["matchId", "teamId"], how="left") | |
| # columnas que definen el "estado a la fecha" de un equipo (lado propio) | |
| self_cols = [c for c in m.columns if c.startswith(("self_", "selfH_", "selfA_"))] | |
| keep = ["Competencia", "Temporada", "teamId", "TeamName", "fecha", "formation"] + self_cols | |
| keep = [c for c in keep if c in m.columns] | |
| # última fila por (liga, temporada, equipo) — su forma más reciente | |
| m = m.sort_values(["Competencia", "Temporada", "teamId", "fecha", "matchId"]) | |
| latest = m[keep].dropna(subset=["self_def_H"]).groupby( | |
| ["Competencia", "Temporada", "teamId"], as_index=False).last() | |
| latest.to_parquet(OUT / "team_features_latest.parquet", index=False) | |
| (OUT / "matchup_feat_cols.json").write_text(json.dumps(feat, ensure_ascii=False)) | |
| print(f"team_features_latest: {len(latest)} equipos-temporada | self_cols={len(self_cols)} | feat={len(feat)}") | |
| print("ligas:", latest['Competencia'].nunique(), "| ej:", | |
| latest[latest.Competencia.eq('Liga Profesional Argentina')]['TeamName'].dropna().unique()[:6].tolist()) | |
| if __name__ == "__main__": | |
| main() | |