"""Shared leaderboard renderer for the Calibration Benchmark dashboard. Extracts the scored-table pipeline from the home page so both the Optimization leaderboard (home) and the UQ leaderboard (pages/UQLeaderboard.py) can reuse it without duplicating code. Usage:: from common.leaderboard import render_leaderboard render_leaderboard( metric_store, target_col="rmse_target", target_label="RMSE Target Level", title="Optimization Leaderboard", state_prefix="opt", default_target=1.1, raw_page="pages/RawData.py", ) """ from __future__ import annotations import altair as alt import pandas as pd import streamlit as st try: from common.method_registry import METHOD_COLORS except ModuleNotFoundError: from src.common.method_registry import METHOD_COLORS _SUITABLE = "#009E73" # Okabe-Ito teal-green (colorblind-safe) _UNSUITABLE = "#C0392B" # dark red _UNTESTED = "#BDBDBD" # gray def _method_color(present_abbrevs: list[str]) -> alt.Color: """Color encoding for the ``abbreviation`` field. The scale domain/range is always the full method registry so a given method keeps the same color across every chart and page. The legend, however, is restricted to *present_abbrevs* so it only lists methods actually plotted in this chart section rather than every known method. """ return alt.Color( "abbreviation:N", title="Method", scale=alt.Scale(domain=list(METHOD_COLORS.keys()), range=list(METHOD_COLORS.values())), legend=alt.Legend(values=sorted(present_abbrevs)), ) # update_type display order in the leaderboard table (lower = earlier). # Unknown/unmapped update types fall back to 99 and appear at the end. # See method_registry.py for the full four-axis taxonomy. _UPDATE_TYPE_ORDER: dict[str, int] = { "kalman": 0, "gradient": 1, "general": 2, } def _render_suitability_table( store: pd.DataFrame, target_col: str, suitability_target: float, benchmark_dims: dict[str, tuple[int, int, int]] | None = None, failure_threshold: float = 20.0, ratio_threshold: float = 3.0, ) -> None: """Render the method × benchmark suitability grid above the leaderboard controls.""" benchmarks = sorted( store["benchmark"].unique().tolist(), key=lambda bm: benchmark_dims[bm][0] if (benchmark_dims and bm in benchmark_dims) else bm, ) _abbr_update_type = ( store[["abbreviation", "update_type"]].dropna() .drop_duplicates("abbreviation") .set_index("abbreviation")["update_type"] .to_dict() ) methods = sorted( store["abbreviation"].dropna().unique().tolist(), key=lambda a: (_UPDATE_TYPE_ORDER.get(_abbr_update_type.get(a, ""), 99), a), ) target_str = str(float(suitability_target)) target_str_col = f"{target_col}_str" target_df = store.copy() target_df[target_str_col] = target_df[target_col].astype(str) target_df = target_df[target_df[target_str_col] == target_str] # Global best per benchmark: min metric across all methods with failure_rate < threshold global_best: dict[str, float | None] = {} for bm in benchmarks: bm_df = target_df[target_df["benchmark"] == bm] qualifying = bm_df[bm_df["failure_rate"] < failure_threshold].dropna(subset=["metric"]) global_best[bm] = float(qualifying["metric"].min()) if not qualifying.empty else None cell_text: dict[str, dict[str, str]] = {} cell_color: dict[str, dict[str, str]] = {} for method in methods: cell_text[method] = {} cell_color[method] = {} for bm in benchmarks: sub = target_df[ (target_df["abbreviation"] == method) & (target_df["benchmark"] == bm) ] if sub.empty: cell_text[method][bm] = "—" cell_color[method][bm] = _UNTESTED continue qualifying = sub[sub["failure_rate"] < failure_threshold].dropna(subset=["metric"]) if qualifying.empty: cell_text[method][bm] = "failed" cell_color[method][bm] = _UNSUITABLE continue method_best = float(qualifying["metric"].min()) gb = global_best.get(bm) ratio = (method_best / gb) if (gb is not None and gb > 0) else 1.0 cell_text[method][bm] = f"{ratio:.1f}×" cell_color[method][bm] = _SUITABLE if ratio <= ratio_threshold else _UNSUITABLE col_labels: dict[str, str] = {} for bm in benchmarks: if benchmark_dims and bm in benchmark_dims: p, s, o = benchmark_dims[bm] col_labels[bm] = f"{bm} (p={p}, s={s}, o={o})" else: col_labels[bm] = bm display_df = pd.DataFrame(cell_text).T.rename(columns=col_labels) color_df = pd.DataFrame(cell_color).T.rename(columns=col_labels) display_df.index.name = "Method" def _style(df: pd.DataFrame) -> pd.DataFrame: result = pd.DataFrame("", index=df.index, columns=df.columns) for row in df.index: for col in df.columns: bg = color_df.loc[row, col] fg = "#212529" if bg == _UNTESTED else "white" result.loc[row, col] = ( f"background-color: {bg}; color: {fg}; " "text-align: center; font-weight: bold" ) return result st.subheader("Method Suitability Overview") st.caption( f"Evaluated at target = {suitability_target}. " f"**Green**: at some ensemble size, failure rate < {failure_threshold:.0f}% " f"and mean budget ≤ {ratio_threshold:.0f}× the best method (ratio shown). " "**Red** (failed): data present but all runs failed to reach the target. " "**Gray (—)**: no data for this benchmark." ) st.dataframe(display_df.style.apply(_style, axis=None), use_container_width=True) def render_leaderboard( metric_store: pd.DataFrame, *, target_col: str, target_label: str, title: str, state_prefix: str, default_target: float, raw_page: str | None = None, show_failure_panel: bool = False, show_scoring_modes: bool = True, canonical_target_levels: list[float] | None = None, budget_store: pd.DataFrame | None = None, benchmark_dims: dict[str, tuple[int, int, int]] | None = None, ) -> None: """Render a scored leaderboard backed by *metric_store*. Parameters ---------- metric_store: DataFrame produced by ``load_metric_store()`` or ``load_uq_store()``. Must contain at least the columns ``benchmark``, ``algorithm_type``, ``abbreviation``, ``Method``, the four taxonomy tag columns (``parallelism``, ``update_type``, ``method_goal``, ``emulator_use``), ``ensemble_size``, ``metric``, ``failure_rate``, and *target_col*. target_col: Name of the target-coordinate column, e.g. ``"rmse_target"`` or ``"uq_target"``. target_label: Human-readable label for the target-level radio control, e.g. ``"RMSE Target Level"`` or ``"UQ Target Level"``. title: Leaderboard section header text. state_prefix: Short string used to namespace ``st.session_state`` keys so multiple leaderboard pages keep independent control state. Use ``"opt"`` for the Optimization leaderboard and ``"uq"`` for the UQ leaderboard. default_target: Target level shown on first load (e.g. ``1.1`` for optimization, ``1.5`` for UQ). raw_page: Optional Streamlit page path for an "Open Raw Data" link shown at the bottom. Pass ``None`` to suppress the link. show_failure_panel: If ``True``, render a grouped-bar failure-rate chart below the main performance chart. show_scoring_modes: If ``True``, show scoring-mode radio controls. canonical_target_levels: If provided, the target-level selector always offers exactly these values (as strings) regardless of what is present in the data. Use this to pin the UQ leaderboard to its fixed set of target-scaling levels even when some have 100 % failure. budget_store: Optional DataFrame produced by ``load_uq_budget_store()``. When provided, an additional "Mean Iterations for Coverage" section is rendered below the main performance chart. benchmark_dims: Optional mapping of benchmark name → (param_dim, state_dim, output_dim) used to annotate column headers in the suitability table. """ st.header(title) if metric_store.empty: st.warning("No metric data found. Expected NetCDF files in `data/` with a `metric` variable.") return _render_suitability_table( metric_store, target_col, default_target, benchmark_dims=benchmark_dims ) st.divider() # Derived column name for the string version of the target coordinate target_str_col = f"{target_col}_str" benchmark_values = sorted(metric_store["benchmark"].unique().tolist()) selected_benchmark = st.selectbox("Benchmark", options=benchmark_values, index=0) filtered = metric_store[metric_store["benchmark"] == selected_benchmark].copy() filtered[target_str_col] = filtered[target_col].astype(str) if canonical_target_levels is not None: target_options = [str(float(t)) for t in canonical_target_levels] else: target_options = sorted( metric_store[target_col].astype(str).unique().tolist() ) scoring_options = [ "Mean Forward Model Runs", "Minimum Forward Model Runs", "Smallest Optimal Ensemble Size", "Custom Blend", ] # Session-state keys namespaced by state_prefix so two leaderboard pages # don't share control state. k_target = f"{state_prefix}_selected_target" k_scoring = f"{state_prefix}_scoring_mode" k_weight = f"{state_prefix}_fwdruns_weight_percent" k_methods = f"{state_prefix}_selected_methods" default_target_str = str(float(default_target)) # Pre-populate session state so the radio widget and the filter agree on first load. if k_target not in st.session_state or st.session_state[k_target] not in target_options: st.session_state[k_target] = ( default_target_str if default_target_str in target_options else target_options[0] ) selected_target = st.session_state[k_target] current_scoring_mode = st.session_state.get(k_scoring, "Mean Forward Model Runs") if current_scoring_mode not in scoring_options: current_scoring_mode = scoring_options[0] current_fwdruns_weight_percent = int(st.session_state.get(k_weight, 80)) current_fwdruns_weight_percent = max(0, min(100, current_fwdruns_weight_percent)) scoring_mode = current_scoring_mode fwdruns_weight = current_fwdruns_weight_percent / 100.0 ensemble_weight = 1.0 - fwdruns_weight # Available methods for the current benchmark selection; used to populate the # multiselect and to prune any stale saved selections when the benchmark changes. available_methods = sorted(filtered["abbreviation"].dropna().unique().tolist()) saved_methods = st.session_state.get(k_methods, available_methods) valid_saved = [m for m in saved_methods if m in available_methods] st.session_state[k_methods] = valid_saved if valid_saved else available_methods def build_scored_table(input_df: pd.DataFrame, add_rank: bool = True) -> pd.DataFrame: ranking_source = input_df[input_df[target_str_col] == selected_target] if ranking_source.empty: return ranking_source # Failure rate from every row (NaN metric rows carry failure_rate=100) failure_agg = ranking_source.groupby( ["algorithm_type", "abbreviation", "Method"], as_index=False ).agg(**{"Mean Failure Rate (%)": ("failure_rate", "mean")}) # Metric stats only from runs that reached the target (non-NaN metric) valid_rows = ranking_source.dropna(subset=["metric"]) if valid_rows.empty: return pd.DataFrame() scored_df = valid_rows.groupby( ["algorithm_type", "abbreviation", "Method"], as_index=False ).agg( **{"Mean Forward Model Runs": ("metric", "mean")}, **{"Minimum Forward Model Runs": ("metric", "min")}, **{"Ensemble Sizes Used": ("ensemble_size", "nunique")}, ) best_per_target = ( valid_rows.sort_values(["algorithm_type", target_str_col, "metric", "ensemble_size"]) .groupby(["algorithm_type", "abbreviation", "Method", target_str_col], as_index=False) .first()[ [ "algorithm_type", "abbreviation", "Method", target_str_col, "ensemble_size", ] ] ) optimal_ensemble = best_per_target.groupby( ["algorithm_type", "abbreviation", "Method"], as_index=False ).agg(**{"Optimal Ensemble Size": ("ensemble_size", "mean")}) scored_df = scored_df.merge( optimal_ensemble, on=["algorithm_type", "abbreviation", "Method"], how="left", ) scored_df = scored_df.merge( failure_agg, on=["algorithm_type", "abbreviation", "Method"], how="left", ) scored_df["Optimal Ensemble Size"] = scored_df["Optimal Ensemble Size"].round(2) scored_df["Mean Forward Model Runs"] = scored_df["Mean Forward Model Runs"].round(4) scored_df["Minimum Forward Model Runs"] = scored_df["Minimum Forward Model Runs"].round(4) scored_df["Mean Failure Rate (%)"] = scored_df["Mean Failure Rate (%)"].round(1) mean_runs_min = scored_df["Mean Forward Model Runs"].min() mean_runs_max = scored_df["Mean Forward Model Runs"].max() if mean_runs_max > mean_runs_min: scored_df["mean_runs_score"] = ( 100.0 * (mean_runs_max - scored_df["Mean Forward Model Runs"]) / (mean_runs_max - mean_runs_min) ) else: scored_df["mean_runs_score"] = 100.0 minimum_runs_min = scored_df["Minimum Forward Model Runs"].min() minimum_runs_max = scored_df["Minimum Forward Model Runs"].max() if minimum_runs_max > minimum_runs_min: scored_df["minimum_runs_score"] = ( 100.0 * (minimum_runs_max - scored_df["Minimum Forward Model Runs"]) / (minimum_runs_max - minimum_runs_min) ) else: scored_df["minimum_runs_score"] = 100.0 ens_min = scored_df["Optimal Ensemble Size"].min() ens_max = scored_df["Optimal Ensemble Size"].max() if ens_max > ens_min: scored_df["ensemble_score"] = ( 100.0 * (ens_max - scored_df["Optimal Ensemble Size"]) / (ens_max - ens_min) ) else: scored_df["ensemble_score"] = 100.0 if scoring_mode == "Mean Forward Model Runs": scored_df["Score"] = scored_df["mean_runs_score"] sort_columns = ["Mean Forward Model Runs", "Optimal Ensemble Size", "abbreviation"] ascending = [True, True, True] elif scoring_mode == "Minimum Forward Model Runs": scored_df["Score"] = scored_df["minimum_runs_score"] sort_columns = ["Minimum Forward Model Runs", "Optimal Ensemble Size", "abbreviation"] ascending = [True, True, True] elif scoring_mode == "Smallest Optimal Ensemble Size": scored_df["Score"] = scored_df["ensemble_score"] sort_columns = ["Optimal Ensemble Size", "Mean Forward Model Runs", "abbreviation"] ascending = [True, True, True] else: scored_df["Score"] = ( fwdruns_weight * scored_df["mean_runs_score"] + ensemble_weight * scored_df["ensemble_score"] ) sort_columns = ["Score", "Mean Forward Model Runs", "Optimal Ensemble Size", "abbreviation"] ascending = [False, True, True, True] scored_df = scored_df.sort_values(sort_columns, ascending=ascending).reset_index(drop=True) if add_rank: scored_df["Rank"] = scored_df.index + 1 scored_df["Placement"] = scored_df["Rank"].apply( lambda rank: f"{ {1: '🥇', 2: '🥈', 3: '🥉'}.get(rank, '')} #{rank}".strip() ) # Append methods that had data for this target but every seed failed (all-NaN metric). # They appear at the bottom of the table as "DNF" so users can distinguish # "tried and failed" from "not tested on this benchmark". tried_abbrevs = set(ranking_source["abbreviation"].dropna().unique()) ranked_abbrevs = set(scored_df["abbreviation"].dropna().unique()) dnf_abbrevs = tried_abbrevs - ranked_abbrevs if dnf_abbrevs: dnf_rows = failure_agg[failure_agg["abbreviation"].isin(dnf_abbrevs)].copy() for col in [ "Score", "mean_runs_score", "minimum_runs_score", "ensemble_score", "Mean Forward Model Runs", "Minimum Forward Model Runs", "Optimal Ensemble Size", "Ensemble Sizes Used", ]: dnf_rows[col] = float("nan") if add_rank: dnf_rows["Rank"] = float("nan") dnf_rows["Placement"] = "DNF" scored_df = pd.concat([scored_df, dnf_rows], ignore_index=True) return scored_df leaderboard_df = build_scored_table(filtered, add_rank=True) # Attach the four taxonomy tags (constant per abbreviation, so this is a # plain lookup merge rather than a groupby key threaded through every # aggregation above). if not leaderboard_df.empty: tag_cols = ["abbreviation", "parallelism", "update_type", "method_goal", "emulator_use"] tag_lookup = filtered[tag_cols].drop_duplicates("abbreviation") leaderboard_df = leaderboard_df.merge(tag_lookup, on="abbreviation", how="left") # Re-sort by update_type group, then by performance within each group. # Placement numbers still reflect overall performance rank. if not leaderboard_df.empty: leaderboard_df = leaderboard_df.assign( _update_type_sort=leaderboard_df["update_type"].map(_UPDATE_TYPE_ORDER).fillna(99), _is_dnf=leaderboard_df["Score"].isna(), ).sort_values( ["_update_type_sort", "_is_dnf", "Score"], ascending=[True, True, False], na_position="last", ).drop(columns=["_update_type_sort", "_is_dnf"]).reset_index(drop=True) if scoring_mode == "Mean Forward Model Runs": score_basis = "mean forward-model runs at the selected target level (lower is better)" elif scoring_mode == "Minimum Forward Model Runs": score_basis = "minimum forward-model runs at the selected target level (lower is better)" elif scoring_mode == "Smallest Optimal Ensemble Size": score_basis = "mean optimal ensemble size at the selected target level (lower is better)" else: score_basis = ( f"weighted blend of normalized forward-model-runs score ({fwdruns_weight:.0%}) " f"and normalized ensemble-size score ({ensemble_weight:.0%})" ) # Controls — always visible (not collapsed behind a dropdown) so users can # change target/scoring/methods even when the current selection yields all # failures. st.subheader("Scoring & Target Controls") st.radio( target_label, options=target_options, horizontal=True, key=k_target, ) if show_scoring_modes: st.radio( "Scoring Method", options=scoring_options, horizontal=True, key=k_scoring, ) if st.session_state.get(k_scoring, "Mean Forward Model Runs") == "Custom Blend": st.slider( "Blend Weight: Forward Runs vs Ensemble Size", min_value=0, max_value=100, step=5, key=k_weight, help=( "Higher forward-runs weight prioritizes fewer model evaluations; " "higher ensemble-size weight prioritizes smaller ensembles." ), ) st.multiselect( "Methods to display in charts", options=available_methods, key=k_methods, ) selected_methods = st.session_state.get(k_methods, available_methods) if not selected_methods: selected_methods = available_methods st.divider() # --- Charts (rendered above the leaderboard table, below the controls) --- if not leaderboard_df.empty: st.subheader("Mean Forward Model Runs vs Ensemble Size") chart_source = filtered[filtered[target_str_col] == selected_target] chart_source = chart_source[chart_source["abbreviation"].isin(selected_methods)] chart_df = chart_source.dropna(subset=["metric"]).groupby( ["abbreviation", "ensemble_size"], as_index=False ).agg(mean_forward_runs=("metric", "mean")) all_ens_combos = chart_source[["abbreviation", "ensemble_size"]].drop_duplicates() ens_ticks = sorted(all_ens_combos["ensemble_size"].unique().tolist()) if not all_ens_combos.empty else [] main_color = _method_color(all_ens_combos["abbreviation"].unique().tolist()) if not chart_df.empty: _ok = chart_df[["abbreviation", "ensemble_size"]].assign(_ok=True) fail_df = all_ens_combos.merge(_ok, on=["abbreviation", "ensemble_size"], how="left") fail_df = fail_df[fail_df["_ok"].isna()].drop(columns="_ok").assign(mean_forward_runs=0.0) else: fail_df = all_ens_combos.assign(mean_forward_runs=0.0) all_failed = chart_df.empty chart_layers = [] if not chart_df.empty: chart_layers.append( alt.Chart(chart_df) .mark_line(point=True) .encode( x=alt.X( "ensemble_size:Q", title="Ensemble Size", axis=alt.Axis(values=ens_ticks, format="d"), ), y=alt.Y("mean_forward_runs:Q", title="Mean Forward Model Runs"), color=main_color, tooltip=["abbreviation", "ensemble_size", alt.Tooltip("mean_forward_runs:Q", format=".4f")], ) ) if not fail_df.empty: y_fwd = ( alt.Y("mean_forward_runs:Q", title="Mean Forward Model Runs", scale=alt.Scale(domain=[0, 1])) if all_failed else alt.Y("mean_forward_runs:Q", title="Mean Forward Model Runs") ) chart_layers.append( alt.Chart(fail_df) .mark_point(shape="cross", angle=45, size=200, filled=True, opacity=1.0) .encode( x=alt.X( "ensemble_size:Q", title="Ensemble Size", axis=alt.Axis(values=ens_ticks, format="d"), ), y=y_fwd, color=main_color, tooltip=[ alt.Tooltip("abbreviation:N", title="Method"), alt.Tooltip("ensemble_size:Q", title="Ensemble Size"), alt.Tooltip("mean_forward_runs:Q", title="Value (all failed)"), ], ) ) # Dashed vertical rule for methods with only one ensemble size so they # remain easy to spot when multi-ensemble methods dominate the x-axis. if not all_ens_combos.empty: single_ens_abbrevs = ( all_ens_combos.groupby("abbreviation")["ensemble_size"] .nunique() .pipe(lambda s: s[s == 1].index.tolist()) ) if single_ens_abbrevs: rule_df = ( all_ens_combos[all_ens_combos["abbreviation"].isin(single_ens_abbrevs)] .drop_duplicates() ) chart_layers.append( alt.Chart(rule_df) .mark_rule(strokeDash=[4, 4], opacity=0.5) .encode( x=alt.X( "ensemble_size:Q", axis=alt.Axis(values=ens_ticks, format="d"), ), color=main_color, tooltip=[ alt.Tooltip("abbreviation:N", title="Method"), alt.Tooltip("ensemble_size:Q", title="Ensemble Size"), ], ) ) if chart_layers: st.altair_chart(alt.layer(*chart_layers), use_container_width=True) # Mean-iterations-for-coverage section (UQ only, when budget_store provided) if budget_store is not None and not budget_store.empty: bf = budget_store[budget_store["benchmark"] == selected_benchmark].copy() bf[target_str_col] = bf[target_col].astype(str) bf = bf[bf[target_str_col] == selected_target] bf = bf[bf["abbreviation"].isin(selected_methods)] iters_df = ( bf[["abbreviation", "ensemble_size", "mean_iters"]] .dropna(subset=["mean_iters"]) .groupby(["abbreviation", "ensemble_size"], as_index=False) .agg(mean_iters=("mean_iters", "mean")) ) all_ens_combos_iters = bf[["abbreviation", "ensemble_size"]].drop_duplicates() if not all_ens_combos_iters.empty: iters_ticks = sorted(all_ens_combos_iters["ensemble_size"].unique().tolist()) iters_color = _method_color(all_ens_combos_iters["abbreviation"].unique().tolist()) if not iters_df.empty: _ok_iters = iters_df[["abbreviation", "ensemble_size"]].assign(_ok=True) fail_df_iters = all_ens_combos_iters.merge(_ok_iters, on=["abbreviation", "ensemble_size"], how="left") fail_df_iters = fail_df_iters[fail_df_iters["_ok"].isna()].drop(columns="_ok").assign(mean_iters=0.0) else: fail_df_iters = all_ens_combos_iters.assign(mean_iters=0.0) all_failed_iters = iters_df.empty st.subheader("Mean Iterations for Coverage vs Ensemble Size") iters_layers = [] if not iters_df.empty: iters_layers.append( alt.Chart(iters_df) .mark_line(point=True) .encode( x=alt.X( "ensemble_size:Q", title="Ensemble Size", axis=alt.Axis(values=iters_ticks, format="d"), ), y=alt.Y("mean_iters:Q", title="Mean Iterations"), color=iters_color, tooltip=[ alt.Tooltip("abbreviation:N", title="Method"), alt.Tooltip("ensemble_size:Q", title="Ensemble Size"), alt.Tooltip("mean_iters:Q", format=".2f", title="Mean Iterations"), ], ) ) if not fail_df_iters.empty: y_iters = ( alt.Y("mean_iters:Q", title="Mean Iterations", scale=alt.Scale(domain=[0, 1])) if all_failed_iters else alt.Y("mean_iters:Q", title="Mean Iterations") ) iters_layers.append( alt.Chart(fail_df_iters) .mark_point(shape="cross", angle=45, size=200, filled=True, opacity=1.0) .encode( x=alt.X( "ensemble_size:Q", title="Ensemble Size", axis=alt.Axis(values=iters_ticks, format="d"), ), y=y_iters, color=iters_color, tooltip=[ alt.Tooltip("abbreviation:N", title="Method"), alt.Tooltip("ensemble_size:Q", title="Ensemble Size"), alt.Tooltip("mean_iters:Q", title="Value (all failed)"), ], ) ) if iters_layers: st.altair_chart(alt.layer(*iters_layers), use_container_width=True) # Failure panel — rendered regardless of whether the scored table has rows if show_failure_panel: failure_source = filtered[filtered[target_str_col] == selected_target] failure_source = failure_source[failure_source["abbreviation"].isin(selected_methods)] if not failure_source.empty: failure_df = failure_source.groupby( ["abbreviation", "ensemble_size"], as_index=False ).agg(mean_failure_rate=("failure_rate", "mean")) failure_df = failure_df.sort_values("ensemble_size") st.subheader(f"Failure Rate of Hitting Target {selected_target}") ens_ticks_fail = sorted(failure_df["ensemble_size"].unique().tolist()) failure_chart = ( alt.Chart(failure_df) .mark_bar() .encode( x=alt.X( "ensemble_size:O", title="Ensemble Size", sort=[str(e) for e in ens_ticks_fail], axis=alt.Axis(labelAngle=0), ), xOffset=alt.XOffset("abbreviation:N"), y=alt.Y( "mean_failure_rate:Q", title="Failure Rate (%)", scale=alt.Scale(domain=[0, 100]), ), color=_method_color(failure_df["abbreviation"].unique().tolist()), tooltip=[ alt.Tooltip("abbreviation:N", title="Method"), alt.Tooltip("ensemble_size:O", title="Ensemble Size"), alt.Tooltip("mean_failure_rate:Q", format=".1f", title="Failure Rate (%)"), ], ) ) ceiling_line = ( alt.Chart(alt.Data(values=[{}])) .mark_rule(color="grey", strokeDash=[4, 4]) .encode(y=alt.datum(100)) ) st.altair_chart(failure_chart + ceiling_line, use_container_width=True) st.divider() # --- Leaderboard table (rendered below the charts) --- table_column_order = [ "Placement", "abbreviation", "Method", "update_type", "parallelism", "method_goal", "emulator_use", "Score", "Mean Forward Model Runs", "Minimum Forward Model Runs", "Mean Failure Rate (%)", "Optimal Ensemble Size", "Ensemble Sizes Used", ] if leaderboard_df.empty: st.warning( "All runs failed to reach the target at this selection. " "See the failure rate chart above." if show_failure_panel else "No rows available for the current benchmark/target selection." ) else: st.subheader(f"Ranked Leaderboard — {selected_benchmark}") st.dataframe( leaderboard_df, hide_index=True, use_container_width=True, column_config={ "Placement": st.column_config.TextColumn("Placement"), "update_type": st.column_config.TextColumn("Update Type"), "parallelism": st.column_config.TextColumn("Parallelism"), "method_goal": st.column_config.TextColumn("Method Goal"), "emulator_use": st.column_config.TextColumn("Emulator Use"), "Method": st.column_config.TextColumn("Method"), "abbreviation": st.column_config.TextColumn("Abbrev."), "Mean Forward Model Runs": st.column_config.NumberColumn("Mean Forward Model Runs", format="%.4f"), "Minimum Forward Model Runs": st.column_config.NumberColumn("Minimum Forward Model Runs", format="%.4f"), "Score": st.column_config.ProgressColumn("Score (0-100)", min_value=0.0, max_value=100.0, format="%.1f"), "Optimal Ensemble Size": st.column_config.NumberColumn("Mean Optimal Ensemble Size", format="%.2f"), "Mean Failure Rate (%)": st.column_config.NumberColumn("Mean Failure Rate (%)", format="%.1f"), "Ensemble Sizes Used": st.column_config.NumberColumn("Ensemble Sizes Used", format="%d"), }, column_order=table_column_order, ) st.info( f"Score is a normalized 0–100 ranking based on **{score_basis}**. " "Values are computed from all ensemble sizes after averaging over random seeds." ) st.caption("Top 3 are shown as podium spots; remaining methods are directly comparable via normalized score.") if raw_page is not None: st.page_link(raw_page, label="Open Raw Data & CSV Export", icon="🧾")