File size: 10,368 Bytes
0173ac0 0769e78 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 10b5ae2 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 38d7b65 0173ac0 0769e78 38d7b65 0173ac0 0769e78 0173ac0 4def061 0173ac0 480be44 069756a 4def061 38d7b65 2de5071 069756a 38d7b65 4def061 069756a 7df207c 4def061 38d7b65 4def061 38d7b65 4def061 38d7b65 10b5ae2 38d7b65 4def061 38d7b65 10b5ae2 38d7b65 10b5ae2 38d7b65 10b5ae2 4def061 10b5ae2 38d7b65 4def061 38d7b65 4def061 10b5ae2 38d7b65 10b5ae2 4def061 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 | import os, io, re, math
import pandas as pd
import numpy as np
import gradio as gr
import pycountry
# =======================
# CONFIG
# =======================
DATA_FILE = "players.dataset.xlsx"
BASE_POSITIONS = ["CB","RB","LB","CDM","CM","CAM","RW","ST","LW"]
EXCLUDED_NATIONS = {
"PALESTINE","STATE OF PALESTINE","PALESTINIAN TERRITORY","PALESTINIAN TERRITORIES","PSE","PS"
}
CANON = {
"name": ["name","player","player_name"],
"age": ["age"],
"position": ["position","pos"],
"nation": ["nation","nationality","country","citizenship"],
"club": ["club","team","current_club"],
"overall": ["overall","rating","ovr"],
"potential": ["potential","pot"],
"value": ["value","value_eur","market_value","market_value_eur"],
"wage": ["wage","wage_eur","salary","salary_eur"],
"height_cm": ["height","height_cm","cm_height"],
"weight_kg": ["weight","weight_kg","kg_weight"],
}
# =======================
# Helpers
# =======================
def _canon_map(columns):
cols = [str(c).strip() for c in columns]
lower = [c.lower().strip() for c in cols]
out = {}
for std, variants in CANON.items():
for v in variants:
if v in lower:
out[std] = cols[lower.index(v)]
break
return out
def _to_number(x):
if pd.isna(x): return np.nan
s = re.sub(r"[^\d.\-]", "", str(x))
try: return float(s)
except: return np.nan
def _to_full_country(n):
if pd.isna(n): return None
s = str(n).strip()
candidate = None
if len(s) <= 3:
c = pycountry.countries.get(alpha_3=s.upper())
if c: candidate = c.name
if candidate is None:
c = pycountry.countries.get(alpha_2=s.upper())
if c: candidate = c.name
if candidate is None:
candidate = s
norm = re.sub(r"\s+"," ", candidate).strip().upper()
if norm in EXCLUDED_NATIONS:
return None
return candidate
def load_df():
if not os.path.exists(DATA_FILE):
return pd.DataFrame(columns=[
"Name","Age","Position","Nation","Club","Overall","Potential",
"Height_cm","Weight_kg","Value","Wage"
])
df = pd.read_excel(DATA_FILE, engine="openpyxl")
df.columns = [str(c).strip() for c in df.columns]
cmap = _canon_map(df.columns)
out = pd.DataFrame()
out["Name"] = df.get(cmap.get("name"), pd.Series(dtype=str)).astype(str).str.strip()
out["Age"] = df.get(cmap.get("age"), pd.Series(dtype=object)).apply(_to_number)
out["Position"] = df.get(cmap.get("position"), pd.Series(dtype=str)).astype(str).str.upper().str.strip()
nat_raw = df.get(cmap.get("nation"), pd.Series(dtype=str))
out["Nation"] = nat_raw.apply(_to_full_country) if nat_raw is not None else pd.Series(dtype=str)
club = df.get(cmap.get("club"), pd.Series(dtype=str)).astype(str).str.strip()
club = club.replace({"": pd.NA, "nan": pd.NA, "None": pd.NA})
out["Club"] = club
out["Overall"] = df.get(cmap.get("overall"), pd.Series(dtype=object)).apply(_to_number)
out["Potential"] = df.get(cmap.get("potential"), pd.Series(dtype=object)).apply(_to_number)
out["Value"] = df.get(cmap.get("value"), pd.Series(dtype=object)).apply(_to_number)
out["Wage"] = df.get(cmap.get("wage"), pd.Series(dtype=object)).apply(_to_number)
out["Height_cm"] = df.get(cmap.get("height_cm"), pd.Series(dtype=object)).apply(_to_number)
out["Weight_kg"] = df.get(cmap.get("weight_kg"), pd.Series(dtype=object)).apply(_to_number)
out = out[~out["Nation"].isna()].reset_index(drop=True)
return out
DF = load_df()
def positions_list():
vals = set(BASE_POSITIONS)
if "Position" in DF.columns:
vals |= set(str(x).upper().strip() for x in DF["Position"].dropna().unique())
return sorted(vals)
def dataset_nations():
if "Nation" not in DF.columns: return []
vals = [n for n in DF["Nation"].dropna().unique().tolist() if str(n).strip()]
return sorted(set(vals))
def clubs_list():
if "Club" not in DF.columns: return []
vals = [c for c in DF["Club"].dropna().unique().tolist() if str(c).strip()]
return sorted(set(vals))
def filter_players(positions, nations, clubs, min_overall, min_potential,
max_age, min_h, max_h, min_w, max_w, max_val, max_wage, query):
df = DF.copy()
if positions: df = df[df["Position"].isin(positions)]
if nations: df = df[df["Nation"].isin(nations)]
if clubs: df = df[df["Club"].isin(clubs)]
if not math.isnan(min_overall): df = df[df["Overall"] >= min_overall]
if not math.isnan(min_potential): df = df[df["Potential"] >= min_potential]
if not math.isnan(max_age): df = df[df["Age"] <= max_age]
if not math.isnan(min_h): df = df[df["Height_cm"] >= min_h]
if not math.isnan(max_h): df = df[df["Height_cm"] <= max_h]
if not math.isnan(min_w): df = df[df["Weight_kg"] >= min_w]
if not math.isnan(max_w): df = df[df["Weight_kg"] <= max_w]
if not math.isnan(max_val): df = df[df["Value"] <= max_val]
if not math.isnan(max_wage): df = df[df["Wage"] <= max_wage]
if query:
q = query.strip().lower()
df = df[df.apply(lambda r: any(
q in str(r.get(c, "")).lower()
for c in ["Name","Club","Position","Nation"]
), axis=1)]
cols = [c for c in ["Name","Age","Position","Nation","Club","Overall","Potential",
"Height_cm","Weight_kg","Value","Wage"] if c in df.columns]
return df[cols].reset_index(drop=True)
def to_csv_bytes(df):
buf = io.StringIO()
df.to_csv(buf, index=False, encoding="utf-8")
return buf.getvalue().encode("utf-8")
# =======================
# UI (clean aesthetic; equal bubble heights; Search in Results header)
# =======================
THEME = gr.themes.Soft(primary_hue="blue")
CSS = """
#title { text-align:center; }
.bubble { background:#fff; border-radius:16px; padding:12px; box-shadow:0 2px 10px rgba(0,0,0,.06); display:flex; flex-direction:column; justify-content:space-between; min-height:120px; }
.bubble.tall { min-height: 220px; }
.grid { gap:12px; }
.stat { font-weight:600; font-size:14px; }
.banner { background:#f6f7ff; border:1px solid #e3e6ff; padding:10px 12px; border-radius:12px; }
.header-row { display:flex; align-items:center; justify-content:space-between; gap:12px; }
.header-row .stats { display:flex; gap:16px; }
"""
with gr.Blocks(theme=THEME, css=CSS) as demo:
# Top banner confirms linkage
if DF.empty:
gr.Markdown("<div class='banner'><b>No players loaded.</b> Make sure <code>players.dataset.xlsx</code> is in the root.</div>")
else:
gr.Markdown(f"<div class='banner'>Loaded <b>{len(DF)}</b> players.</div>")
gr.Markdown("<h1 id='title'>ProScout β Player Finder</h1>")
with gr.Row():
# Column 1 β categorical + search (each in its own bubble)
with gr.Column(scale=1, elem_classes="grid"):
with gr.Group(elem_classes="bubble tall"):
pos = gr.CheckboxGroup(positions_list(), label="Positions", value=[], info="Pick one or more")
with gr.Group(elem_classes="bubble"):
nat = gr.Dropdown(dataset_nations(), multiselect=True, label="Nations", value=[], filterable=True)
with gr.Group(elem_classes="bubble"):
clu = gr.Dropdown(clubs_list(), multiselect=True, label="Clubs", value=[], filterable=True)
with gr.Group(elem_classes="bubble"):
query = gr.Textbox(label="Search", placeholder="e.g., player name, club, role")
# Column 2 β ratings & age
with gr.Column(scale=1, elem_classes="grid"):
with gr.Group(elem_classes="bubble"):
min_ovr = gr.Slider(0, 99, value=0, step=1, label="Min Overall")
with gr.Group(elem_classes="bubble"):
min_pot = gr.Slider(0, 99, value=0, step=1, label="Min Potential")
with gr.Group(elem_classes="bubble"):
max_age = gr.Slider(15, 45, value=45, step=1, label="Max Age")
# Column 3 β physical & financial
with gr.Column(scale=1, elem_classes="grid"):
with gr.Group(elem_classes="bubble"):
min_h = gr.Slider(140, 210, value=140, step=1, label="Min Height (cm)")
with gr.Group(elem_classes="bubble"):
max_h = gr.Slider(140, 210, value=210, step=1, label="Max Height (cm)")
with gr.Group(elem_classes="bubble"):
min_w = gr.Slider(45, 120, value=45, step=1, label="Min Weight (kg)")
with gr.Group(elem_classes="bubble"):
max_w = gr.Slider(45, 120, value=120, step=1, label="Max Weight (kg)")
with gr.Group(elem_classes="bubble"):
max_val = gr.Number(value=np.nan, label="Max Value (EUR)")
with gr.Group(elem_classes="bubble"):
max_wage = gr.Number(value=np.nan, label="Max Wage (EUR)")
# Results panel β Search button in header next to stats
with gr.Column(scale=2):
with gr.Group(elem_classes="bubble tall"):
header = gr.HTML("<div class='header-row'><div class='stats'><span id='stat-count'></span><span id='stat-ovr'></span><span id='stat-age'></span></div></div>")
results = gr.Dataframe(row_count=(12,"dynamic"), wrap=True, interactive=False, label="Results")
count_box = gr.Markdown("", elem_id="stat-count")
avg_ovr_box = gr.Markdown("", elem_id="stat-ovr")
avg_age_box = gr.Markdown("", elem_id="stat-age")
btn = gr.Button("Search", variant="primary")
# Main callback
def _run(positions, nations, clubs, min_overall, min_potential,
max_age, min_h, max_h, min_w, max_w, max_val, max_wage, query):
df = filter_players(positions, nations, clubs, min_overall, min_potential,
max_age, min_h, max_h, min_w, max_w, max_val, max_wage, query)
count = len(df)
avg_ovr = round(df["Overall"].mean(), 2) if "Overall" in df and not df["Overall"].isna().all() else "-"
avg_age = round(df["Age"].mean(), 2) if "Age" in df and not df["Age"].isna(_
|