import streamlit as st from pathlib import Path st.set_page_config(page_title="Calibration Benchmark", page_icon="๐Ÿ ", layout="wide") # Sidebar navigation st.sidebar.title("Navigation") st.sidebar.page_link("streamlit_app.py", label="Home", icon="๐Ÿ ") st.sidebar.page_link("pages/OptimizationLeaderboard.py", label="Optimization Leaderboard", icon="๐Ÿ“Š") st.sidebar.page_link("pages/UQLeaderboard.py", label="UQ Leaderboard", icon="๐ŸŽฏ") st.sidebar.page_link("pages/MethodDetails.py", label="Methods", icon="๐Ÿ“˜") st.sidebar.page_link("pages/RawData.py", label="Get Data", icon="๐Ÿงพ") st.title("Calibration Benchmark") st.markdown( "A benchmark comparing parameter-calibration methods on chaotic dynamical systems. " "Methods are ranked by **forward-model run efficiency** โ€” how many forward-model " "evaluations are needed, on average across random seeds, to reach a target accuracy." ) st.divider() col1, col2 = st.columns(2, gap="large") with col1: st.subheader("๐Ÿ“Š Optimization Leaderboard") st.markdown( "Ranks methods by mean forward-model runs to reach an **RMSE target** on " "Lorenz-63 and Lorenz-96 benchmarks. " "Lower is better; failed runs are tracked separately as a failure rate." ) st.page_link("pages/OptimizationLeaderboard.py", label="Go to Optimization Leaderboard โ†’") with col2: st.subheader("๐ŸŽฏ UQ Leaderboard") st.markdown( "Ranks methods by mean forward-model runs to reach an **uncertainty quantification " "target**. Same metric and benchmarks as the Optimization Leaderboard, evaluated " "at a UQ-specific convergence criterion." ) st.page_link("pages/UQLeaderboard.py", label="Go to UQ Leaderboard โ†’") st.divider() st.subheader("Benchmarks") st.markdown( "Results are reported on four benchmark configurations of the [Lorenz system]" "(https://en.wikipedia.org/wiki/Lorenz_system), a standard testbed for " "data-assimilation and calibration algorithms:" ) _media = Path(__file__).parent / "media" st.markdown("- **L63** โ€” Lorenz-63 โ€” 3-variable chaotic attractor; learn 2 parameters, strongly nonlinear.") st.image( str(_media / "posterior_ribbons_20_13_k5.png"), caption="Example prior-posterior & truth. L: Difference to true parameter. R: data-sample/output predictived distribution (state-mean [1:3], state-covariance (diag [4:6] and off-diag [7:9]))", width=800, ) st.markdown("- **L96** โ€” Lorenz-96 (40-variable); learn 1-parameter constant forcing.") st.image( str(_media / "posterior_ribbons_const-force_12_1_k3.png"), caption="Example prior-posterior & truth. L: Difference to true parameter. C: parameter-induced forcing of the L96 system. R: data-sample/output predictived distribution ([1:40] state-mean [41:80] state-std)", width=1200, ) st.markdown("- **L96_SPATIAL_FORCING** โ€” Lorenz-96 (40-variable) with spatially-varying forcing; learn 40 parameters; moderately correlated prior.") st.image( str(_media / "posterior_ribbons_vec-force_65_1_k3.png"), caption="Example prior-posterior & truth. L: Difference to true parameters. C: parameter-induced forcing of the L96 system. R: data-sample/output predictived distribution ([1:40] state-mean [41:80] state-std)", width=1200, ) st.markdown("- **L96_NN_FORCING** โ€” Lorenz-96 (100-variable) with a neural-network forcing; Learn 61 parameters (weights and biases of the network). Reasonable prior given.") st.image( str(_media / "posterior_ribbons_flux-force_80_1_k3.png"), caption="Example prior-posterior & truth. L: Difference to true parameters (weights). C: parameter-induced forcing of the L96 system. R: data-sample/output predictived distribution ([1:100] state-mean [101:200] state-std)", width=1200, ) st.subheader("Method families") st.markdown( "Methods are grouped into three families. See the **๐Ÿ“˜ Methods** page for " "citations and per-method performance charts." ) families = { "Kalman": "Ensemble Kalman variants (TEKI, ETKI, IEKF, UKI) โ€” update an ensemble of parameter guesses via a linearized observation operator.", "Bayesian": "Sampling-based approaches (ABC, HM) โ€” explore parameter space without requiring gradient information.", "Calibrate-then-emulate": "Two-stage pipelines (CES-EKI-DMC) โ€” use an initial calibration phase to build a cheap emulator, then sample the posterior via MCMC.", } for family, desc in families.items(): st.markdown(f"- **{family}** โ€” {desc}") st.subheader("Key metric") st.markdown( "The reported metric is the **mean number of forward-model evaluations** required " "to reach the target, averaged over random seeds. " "A value of **โˆ’1** indicates a failed run (did not reach the target); " "the **failure rate** shows the fraction of seeds that failed." )