from __future__ import annotations import html import math from dataclasses import dataclass from pathlib import Path from typing import Iterable import pandas as pd DATA_PATH = Path(__file__).parent / "data" / "leaderboard.csv" METRIC_COLUMNS = [ "Vis", "Aud (PQ)", "AV", "Lip", "Text", "Face", "Music", "Speech", "Lo-Phy", "Hi-Phy", "Holistic", "Total", ] SORT_COLUMNS = ["Rank", "Model", "Components", "Component Type", *METRIC_COLUMNS] SORT_CHOICES = [ ("Rank", "Rank"), ("Model", "Model"), ("Components", "Components"), ("Type", "Component Type"), *[(metric, metric) for metric in METRIC_COLUMNS], ] LOWER_IS_BETTER = {"AV", "Lip"} NUMERIC_COLUMNS = METRIC_COLUMNS FORMATTERS = { "Vis": "{:.3f}", "Aud (PQ)": "{:.2f}", "AV": "{:.2f}", "Lip": "{:.2f}", "Text": "{:.2f}", "Face": "{:.2f}", "Music": "{:.2f}", "Speech": "{:.2f}", "Lo-Phy": "{:.2f}", "Hi-Phy": "{:.2f}", "Holistic": "{:.2f}", "Total": "{:.2f}", } GROUP_WEIGHTS = { "Basic Uni-modal": 0.2, "Basic Cross-modal": 0.2, "Fine-grained": 0.6, } GROUP_DIMENSIONS = { "Basic Uni-modal": ["Vis", "Aud (PQ)"], "Basic Cross-modal": ["AV", "Lip"], "Fine-grained": ["Text", "Face", "Music", "Speech", "Lo-Phy", "Hi-Phy", "Holistic"], } @dataclass(frozen=True) class Standing: best: float | None second: float | None def load_leaderboard(path: Path = DATA_PATH) -> pd.DataFrame: df = pd.read_csv(path) for column in NUMERIC_COLUMNS: df[column] = pd.to_numeric(df[column], errors="coerce") df = df.sort_values("Total", ascending=False, na_position="last").reset_index(drop=True) df.insert(0, "Rank", range(1, len(df) + 1)) return df def filter_leaderboard( df: pd.DataFrame, component_type: str = "All", query: str = "", sort_by: str = "Total", sort_order: str = "Descending", ) -> pd.DataFrame: view = df.copy() if component_type != "All": view = view[view["Component Type"] == component_type] query = query.strip().lower() if query: mask = ( view["Model"].str.lower().str.contains(query, regex=False) | view["Components"].str.lower().str.contains(query, regex=False) ) view = view[mask] sort_by = _normalize_sort_column(sort_by) ascending = _is_ascending_sort(sort_by, sort_order) if sort_by in view.columns: sort_kwargs = { "ascending": ascending, "na_position": "last", "kind": "mergesort", } if sort_by in {"Model", "Components", "Component Type"}: sort_kwargs["key"] = lambda column: column.astype(str).str.casefold() view = view.sort_values(sort_by, **sort_kwargs) return view.reset_index(drop=True) def metric_standings(df: pd.DataFrame) -> dict[str, Standing]: standings: dict[str, Standing] = {} for metric in METRIC_COLUMNS: values = sorted( {float(v) for v in df[metric].dropna()}, reverse=metric not in LOWER_IS_BETTER, ) standings[metric] = Standing( best=values[0] if values else None, second=values[1] if len(values) > 1 else None, ) return standings def normalized_score(metric: str, value: float) -> float: if metric == "Vis": return _clamp(value * 100.0, 0.0, 100.0) if metric == "Aud (PQ)": return _clamp(value * 10.0, 0.0, 100.0) if metric == "AV": return _clamp(100.0 * (1.0 - value / 0.5), 0.0, 100.0) if metric == "Lip": return _clamp(100.0 * (1.0 - value / 8.0), 0.0, 100.0) if metric == "Lo-Phy": return _clamp(value * 20.0, 0.0, 100.0) return _clamp(value, 0.0, 100.0) def compute_total_from_metrics(row: pd.Series) -> float: group_scores: list[float] = [] group_weights: list[float] = [] for group_name, metrics in GROUP_DIMENSIONS.items(): values = [] for metric in metrics: value = row.get(metric) if pd.isna(value): continue values.append(normalized_score(metric, float(value))) if values: group_scores.append(sum(values) / len(values)) group_weights.append(GROUP_WEIGHTS[group_name]) if not group_scores: return float("nan") weighted = sum(score * weight for score, weight in zip(group_scores, group_weights)) return weighted / sum(group_weights) def render_summary(df: pd.DataFrame, view: pd.DataFrame) -> str: top = df.sort_values("Total", ascending=False).iloc[0] open_source = df[df["Component Type"] == "Open-source"].sort_values("Total", ascending=False) best_open = open_source.iloc[0] if len(open_source) else None best_av = df.sort_values("AV", ascending=True).iloc[0] best_speech = df.sort_values("Speech", ascending=False).iloc[0] cards = [ _summary_card("Models", f"{len(view)} / {len(df)}", "shown in current view"), _summary_card("Top Total", _score(top["Total"]), str(top["Model"])), _summary_card( "Best Open-source", _score(best_open["Total"]) if best_open is not None else "NA", str(best_open["Model"]) if best_open is not None else "No entry", ), _summary_card("Lowest AV Offset", _score(best_av["AV"]), str(best_av["Model"])), _summary_card("Highest Speech", _score(best_speech["Speech"]), str(best_speech["Model"])), ] return '
' + "".join(cards) + "
" def render_table( df: pd.DataFrame, standings: dict[str, Standing], sort_by: str = "Total", sort_order: str = "Descending", ) -> str: if df.empty: return '
No matching models.
' sort_by = _normalize_sort_column(sort_by) headers = [ ("Rank", "Rank"), ("Model", "Model"), ("Components", "Components"), ("Type", "Component Type"), *[(metric, metric) for metric in METRIC_COLUMNS], ] header_html = "".join(_header_cell(label, column, sort_by, sort_order) for label, column in headers) rows = [] for _, row in df.iterrows(): cells = [ f'#{int(row["Rank"])}', f'{html.escape(str(row["Model"]))}', f'{render_component_badges(str(row["Components"]))}', f'{_type_badge(str(row["Component Type"]))}', ] for metric in METRIC_COLUMNS: cells.append(_metric_cell(metric, row[metric], standings[metric])) rows.append("" + "".join(cells) + "") return ( '
' f"{header_html}{''.join(rows)}" "
" ) def render_component_badges(components: str) -> str: badges = [] for component in _split_components(components): lowered = component.lower() kind = "proprietary" if "proprietary" in lowered else "open" if "open-source" in lowered else "neutral" label = component.replace(" (Proprietary)", "").replace(" (Open-source)", "") badges.append(f'{html.escape(label)}') return "".join(badges) def render_profile(df: pd.DataFrame, model: str) -> str: if df.empty: return "" if not model or model not in set(df["Model"]): model = str(df.sort_values("Total", ascending=False).iloc[0]["Model"]) row = df[df["Model"] == model].iloc[0] metric_blocks = [] for metric in METRIC_COLUMNS[:-1]: value = float(row[metric]) normalized = normalized_score(metric, value) direction = "lower is better" if metric in LOWER_IS_BETTER else "higher is better" metric_blocks.append( f"""
{html.escape(metric)} {_format_metric(metric, value)}
{normalized:.1f} normalized, {direction}
""" ) return f"""

Model profile

{html.escape(str(row["Model"]))}

{render_component_badges(str(row["Components"]))}
Total {_score(row["Total"])}
{''.join(metric_blocks)}
""" def render_methodology() -> str: groups = "".join( f"""

{html.escape(name)}

{weight:.1f}

{html.escape(', '.join(metrics))}

""" for name, weight in GROUP_WEIGHTS.items() for metrics in [GROUP_DIMENSIONS[name]] ) return f"""
{groups}

Total uses AVGen-Bench Scheme 2: group-weighted normalized metrics with Vis x 100, Aud(PQ) x 10, Lo-Phy x 20, AV = 100 * max(0, 1 - AV / 0.5), Lip = 100 * max(0, 1 - Lip / 8), and the remaining metrics already on a 0-100 scale.

""" def model_choices(df: pd.DataFrame) -> list[str]: return list(df.sort_values("Total", ascending=False)["Model"]) def _metric_cell(metric: str, value: float, standing: Standing) -> str: if pd.isna(value): return 'NA' numeric = float(value) classes = ["metric-cell"] if _close(numeric, standing.best): classes.append("best") elif _close(numeric, standing.second): classes.append("second") return f'{_format_metric(metric, numeric)}' def _type_badge(component_type: str) -> str: kind = component_type.lower().replace("-", "").replace(" ", "") return f'{html.escape(component_type)}' def _summary_card(label: str, value: str, detail: str) -> str: return f"""
{html.escape(label)} {html.escape(value)} {html.escape(detail)}
""" def _header_cell(label: str, column: str, sort_by: str, sort_order: str) -> str: if column != sort_by: return f"{html.escape(label)}" direction = "ascending" if _is_ascending_sort(sort_by, sort_order) else "descending" indicator = "↑" if direction == "ascending" else "↓" return ( f'' f"{html.escape(label)}" f'' "" ) def _normalize_sort_column(sort_by: str) -> str: if sort_by == "Type": return "Component Type" return sort_by if sort_by in SORT_COLUMNS else "Total" def _is_ascending_sort(sort_by: str, sort_order: str) -> bool: if sort_order == "Best first": return sort_by in LOWER_IS_BETTER return sort_order == "Ascending" def _split_components(value: str) -> Iterable[str]: return [part.strip() for part in value.split("|") if part.strip()] def _format_metric(metric: str, value: float) -> str: return FORMATTERS[metric].format(float(value)) def _score(value: float) -> str: if pd.isna(value): return "NA" return f"{float(value):.2f}" def _close(left: float, right: float | None) -> bool: if right is None: return False return math.isclose(float(left), float(right), rel_tol=0.0, abs_tol=1e-9) def _clamp(value: float, low: float, high: float) -> float: return max(low, min(high, value))