import altair as alt import importlib import pandas as pd import streamlit as st st.set_page_config(layout="wide") if __package__: data_store = importlib.import_module(f"{__package__}.data_store") else: data_store = importlib.import_module("data_store") load_metric_store = data_store.load_metric_store st.title("Calibration Benchmark") # Sidebar navigation st.sidebar.title("Navigation") st.sidebar.page_link("streamlit_app.py", label="Home", icon="๐Ÿ ") st.sidebar.page_link("pages/MethodDetails.py", label="Methods", icon="๐Ÿ“˜") st.sidebar.page_link("pages/RawData.py", label="Get Data", icon="๐Ÿงพ") show_home = st.session_state.get("show_home", True) if show_home: st.header("Calibration Leaderboard") metric_store = load_metric_store() if metric_store.empty: st.warning("No metric data found. Expected NetCDF files in `data/` with a `metric` variable.") st.stop() benchmark_values = sorted(metric_store["benchmark"].unique().tolist()) benchmark_options = ["All"] + benchmark_values default_benchmark_index = 0 selected_benchmark = st.selectbox("Benchmark", options=benchmark_options, index=default_benchmark_index) filtered = metric_store.copy() if selected_benchmark == "All" else metric_store[metric_store["benchmark"] == selected_benchmark].copy() filtered["rmse_target_str"] = filtered["rmse_target"].astype(str) target_options = ["All targets"] + sorted(metric_store["rmse_target"].astype(str).unique().tolist()) scoring_options = [ "Mean Forward Model Runs", "Minimum Forward Model Runs", "Smallest Optimal Ensemble Size", "Custom Blend", ] current_target = st.session_state.get("selected_target", "All targets") if current_target not in target_options: current_target = target_options[0] current_scoring_mode = st.session_state.get("scoring_mode", "Mean Forward Model Runs") if current_scoring_mode not in scoring_options: current_scoring_mode = scoring_options[0] current_fwdruns_weight_percent = int(st.session_state.get("fwdruns_weight_percent", 80)) current_fwdruns_weight_percent = max(0, min(100, current_fwdruns_weight_percent)) selected_target = current_target scoring_mode = current_scoring_mode fwdruns_weight = current_fwdruns_weight_percent / 100.0 ensemble_weight = 1.0 - fwdruns_weight def build_scored_table(input_df, add_rank=True): ranking_source = input_df if selected_target == "All targets" else input_df[input_df["rmse_target_str"] == selected_target] if ranking_source.empty: return ranking_source scored_df = ranking_source.groupby(["algorithm_type", "abbreviation", "Method", "family"], as_index=False).agg( **{"Mean Forward Model Runs": ("metric", "mean")}, **{"Minimum Forward Model Runs": ("metric", "min")}, **{"Targets Used": ("rmse_target_str", "nunique")}, **{"Ensemble Sizes Used": ("ensemble_size", "nunique")}, ) best_per_target = ( ranking_source.sort_values(["algorithm_type", "rmse_target_str", "metric", "ensemble_size"]) .groupby(["algorithm_type", "abbreviation", "Method", "family", "rmse_target_str"], as_index=False) .first()[ [ "algorithm_type", "abbreviation", "Method", "family", "rmse_target_str", "ensemble_size", ] ] ) optimal_ensemble = best_per_target.groupby( ["algorithm_type", "abbreviation", "Method", "family"], as_index=False ).agg(**{"Optimal Ensemble Size": ("ensemble_size", "mean")}) scored_df = scored_df.merge( optimal_ensemble, on=["algorithm_type", "abbreviation", "Method", "family"], how="left", ) scored_df["Optimal Ensemble Size"] = scored_df["Optimal Ensemble Size"].round(2) scored_df["Mean Forward Model Runs"] = scored_df["Mean Forward Model Runs"].round(4) scored_df["Minimum Forward Model Runs"] = scored_df["Minimum Forward Model Runs"].round(4) mean_runs_min = scored_df["Mean Forward Model Runs"].min() mean_runs_max = scored_df["Mean Forward Model Runs"].max() if mean_runs_max > mean_runs_min: scored_df["mean_runs_score"] = 100.0 * (mean_runs_max - scored_df["Mean Forward Model Runs"]) / (mean_runs_max - mean_runs_min) else: scored_df["mean_runs_score"] = 100.0 minimum_runs_min = scored_df["Minimum Forward Model Runs"].min() minimum_runs_max = scored_df["Minimum Forward Model Runs"].max() if minimum_runs_max > minimum_runs_min: scored_df["minimum_runs_score"] = 100.0 * (minimum_runs_max - scored_df["Minimum Forward Model Runs"]) / (minimum_runs_max - minimum_runs_min) else: scored_df["minimum_runs_score"] = 100.0 ens_min = scored_df["Optimal Ensemble Size"].min() ens_max = scored_df["Optimal Ensemble Size"].max() if ens_max > ens_min: scored_df["ensemble_score"] = 100.0 * (ens_max - scored_df["Optimal Ensemble Size"]) / (ens_max - ens_min) else: scored_df["ensemble_score"] = 100.0 if scoring_mode == "Mean Forward Model Runs": scored_df["Score"] = scored_df["mean_runs_score"] sort_columns = ["Mean Forward Model Runs", "Optimal Ensemble Size", "abbreviation"] ascending = [True, True, True] elif scoring_mode == "Minimum Forward Model Runs": scored_df["Score"] = scored_df["minimum_runs_score"] sort_columns = ["Minimum Forward Model Runs", "Optimal Ensemble Size", "abbreviation"] ascending = [True, True, True] elif scoring_mode == "Smallest Optimal Ensemble Size": scored_df["Score"] = scored_df["ensemble_score"] sort_columns = ["Optimal Ensemble Size", "Mean Forward Model Runs", "abbreviation"] ascending = [True, True, True] else: scored_df["Score"] = ( fwdruns_weight * scored_df["mean_runs_score"] + ensemble_weight * scored_df["ensemble_score"] ) sort_columns = ["Score", "Mean Forward Model Runs", "Optimal Ensemble Size", "abbreviation"] ascending = [False, True, True, True] scored_df = scored_df.sort_values(sort_columns, ascending=ascending).reset_index(drop=True) if add_rank: scored_df["Rank"] = scored_df.index + 1 scored_df["Placement"] = scored_df["Rank"].apply( lambda rank: f"{ {1: '๐Ÿฅ‡', 2: '๐Ÿฅˆ', 3: '๐Ÿฅ‰'}.get(rank, '')} #{rank}".strip() ) return scored_df if selected_benchmark == "All": benchmark_scores = [] for benchmark_name in benchmark_values: benchmark_df = metric_store[metric_store["benchmark"] == benchmark_name].copy() benchmark_df["rmse_target_str"] = benchmark_df["rmse_target"].astype(str) scored = build_scored_table(benchmark_df, add_rank=False) if scored.empty: continue scored["benchmark"] = benchmark_name benchmark_scores.append(scored) if benchmark_scores: combined_scores = benchmark_scores[0].copy() if len(benchmark_scores) == 1 else pd.concat(benchmark_scores, ignore_index=True) leaderboard_df = combined_scores.groupby(["algorithm_type", "abbreviation", "Method", "family"], as_index=False).agg( Score=("Score", "mean"), **{"Mean Forward Model Runs": ("Mean Forward Model Runs", "mean")}, **{"Minimum Forward Model Runs": ("Minimum Forward Model Runs", "mean")}, **{"Optimal Ensemble Size": ("Optimal Ensemble Size", "mean")}, **{"Targets Used": ("Targets Used", "mean")}, **{"Ensemble Sizes Used": ("Ensemble Sizes Used", "mean")}, **{"Benchmarks Used": ("benchmark", "nunique")}, ) leaderboard_df["Mean Forward Model Runs"] = leaderboard_df["Mean Forward Model Runs"].round(4) leaderboard_df["Minimum Forward Model Runs"] = leaderboard_df["Minimum Forward Model Runs"].round(4) leaderboard_df["Optimal Ensemble Size"] = leaderboard_df["Optimal Ensemble Size"].round(2) leaderboard_df["Targets Used"] = leaderboard_df["Targets Used"].round().astype(int) leaderboard_df["Ensemble Sizes Used"] = leaderboard_df["Ensemble Sizes Used"].round().astype(int) leaderboard_df = leaderboard_df.sort_values(["Score", "Mean Forward Model Runs", "abbreviation"], ascending=[False, True, True]).reset_index(drop=True) leaderboard_df["Rank"] = leaderboard_df.index + 1 leaderboard_df["Placement"] = leaderboard_df["Rank"].apply( lambda rank: f"{ {1: '๐Ÿฅ‡', 2: '๐Ÿฅˆ', 3: '๐Ÿฅ‰'}.get(rank, '')} #{rank}".strip() ) else: leaderboard_df = pd.DataFrame( columns=[ "Placement", "abbreviation", "Method", "family", "Mean Forward Model Runs", "Minimum Forward Model Runs", "Score", "Optimal Ensemble Size", "Targets Used", "Ensemble Sizes Used", "Benchmarks Used", ] ) else: leaderboard_df = build_scored_table(filtered, add_rank=True) leaderboard_df["Benchmarks Used"] = 1 if scoring_mode == "Mean Forward Model Runs": score_basis = "normalized mean of best forward model runs over selected target levels (lower is better)" elif scoring_mode == "Minimum Forward Model Runs": score_basis = "normalized minimum of forward model runs over selected targets and ensemble sizes (lower is better)" elif scoring_mode == "Smallest Optimal Ensemble Size": score_basis = "normalized mean optimal ensemble size over selected target levels (lower is better)" else: score_basis = ( "weighted blend of normalized forward-model-runs score and normalized ensemble-size score " f"(forward-runs weight {fwdruns_weight:.0%}, ensemble-size weight {ensemble_weight:.0%})" ) if selected_benchmark == "All": score_basis = f"{score_basis}; in All mode, each method's final score is the mean of its per-benchmark scores" if leaderboard_df.empty: st.warning("No rows available for the current benchmark/target selection.") st.stop() if selected_benchmark == "All": table_column_order = [ "Placement", "abbreviation", "Method", "family", "Score", "Targets Used", "Ensemble Sizes Used", "Benchmarks Used", ] else: table_column_order = [ "Placement", "abbreviation", "Method", "family", "Score", "Mean Forward Model Runs", "Minimum Forward Model Runs", "Optimal Ensemble Size", "Targets Used", "Ensemble Sizes Used", "Benchmarks Used", ] st.subheader(f"Ranked Leaderboard โ€” {selected_benchmark}") st.dataframe( leaderboard_df, hide_index=True, width="stretch", column_config={ "Placement": st.column_config.TextColumn("Placement"), "family": st.column_config.TextColumn("Family"), "Method": st.column_config.TextColumn("Method"), "abbreviation": st.column_config.TextColumn("Abbrev."), "Mean Forward Model Runs": st.column_config.NumberColumn("Mean Forward Model Runs", format="%.4f"), "Minimum Forward Model Runs": st.column_config.NumberColumn("Minimum Forward Model Runs", format="%.4f"), "Score": st.column_config.ProgressColumn("Score (0-100)", min_value=0.0, max_value=100.0, format="%.1f"), "Optimal Ensemble Size": st.column_config.NumberColumn("Mean Optimal Ensemble Size", format="%.2f"), "Targets Used": st.column_config.NumberColumn("Targets Used", format="%d"), "Ensemble Sizes Used": st.column_config.NumberColumn("Ensemble Sizes Used", format="%d"), "Benchmarks Used": st.column_config.NumberColumn("Benchmarks Used", format="%d"), }, column_order=table_column_order, ) with st.expander("Scoring & Target Controls", expanded=False): st.radio( "RMSE Target Level", options=target_options, horizontal=True, key="selected_target", ) st.radio( "Scoring Method", options=scoring_options, horizontal=True, key="scoring_mode", ) if st.session_state.get("scoring_mode", "Mean Forward Model Runs") == "Custom Blend": st.slider( "Blend Weight: Forward Runs vs Ensemble Size", min_value=0, max_value=100, step=5, key="fwdruns_weight_percent", help="Higher forward-runs weight prioritizes fewer model evaluations; higher ensemble-size weight prioritizes smaller ensembles.", ) st.info( f"Score is a normalized 0โ€“100 ranking based on **{score_basis}**. " "For Mean/Minimum forward-runs scoring, values are computed from all selected metric target levels and all ensemble sizes " "after averaging over random seeds." ) if selected_benchmark != "All": st.subheader("Mean Forward Model Runs vs Ensemble Size") chart_source = filtered if selected_target == "All targets" else filtered[filtered["rmse_target_str"] == selected_target] chart_df = chart_source.groupby(["abbreviation", "ensemble_size"], as_index=False).agg(mean_forward_runs=("metric", "mean")) chart = ( alt.Chart(chart_df) .mark_line(point=True) .encode( x=alt.X("ensemble_size:Q", title="Ensemble Size"), y=alt.Y("mean_forward_runs:Q", title="Mean Forward Model Runs"), color=alt.Color("abbreviation:N", title="Method"), tooltip=["abbreviation", "ensemble_size", alt.Tooltip("mean_forward_runs:Q", format=".4f")], ) ) st.altair_chart(chart, width="stretch") st.caption("Top 3 are shown as podium spots; remaining methods are directly comparable via normalized score.") st.page_link("pages/RawData.py", label="Open Raw Data & CSV Export", icon="๐Ÿงพ")