import streamlit as st from pathlib import Path st.set_page_config(page_title="Calibration Benchmark", page_icon="๐Ÿ ", layout="wide") # Sidebar navigation st.sidebar.title("Navigation") st.sidebar.page_link("streamlit_app.py", label="Home", icon="๐Ÿ ") st.sidebar.page_link("pages/OptimizationLeaderboard.py", label="Optimization Leaderboard", icon="๐Ÿ“Š") st.sidebar.page_link("pages/UQLeaderboard.py", label="UQ Leaderboard", icon="๐ŸŽฏ") st.sidebar.page_link("pages/MethodDetails.py", label="Methods", icon="๐Ÿ“˜") st.sidebar.page_link("pages/RawData.py", label="Get Data", icon="๐Ÿงพ") st.title("Calibration Benchmark") st.markdown( "A benchmark comparing parameter-calibration methods on chaotic dynamical systems. " "Methods are ranked by **forward-model run efficiency** โ€” how many forward-model " "evaluations are needed, on average across random seeds, to reach a target accuracy." ) st.divider() col1, col2 = st.columns(2, gap="large") with col1: st.subheader("๐Ÿ“Š Optimization Leaderboard") st.markdown( "Ranks methods by mean forward-model runs to reach an **RMSE target** on " "Lorenz-63 and Lorenz-96 benchmarks. " "Lower is better; failed runs are tracked separately as a failure rate." ) st.page_link("pages/OptimizationLeaderboard.py", label="Go to Optimization Leaderboard โ†’") with col2: st.subheader("๐ŸŽฏ UQ Leaderboard") st.markdown( "Ranks methods by mean forward-model runs to reach an **uncertainty quantification " "target**. Same metric and benchmarks as the Optimization Leaderboard, evaluated " "at a UQ-specific convergence criterion." ) st.page_link("pages/UQLeaderboard.py", label="Go to UQ Leaderboard โ†’") st.divider() st.subheader("Benchmarks") st.markdown( "Results are reported on four benchmark configurations of the [Lorenz system]" "(https://en.wikipedia.org/wiki/Lorenz_system), a standard testbed for " "data-assimilation and calibration algorithms:" ) _media = Path(__file__).parent / "media" st.markdown("- **L63** โ€” Lorenz-63 โ€” 3-variable chaotic attractor; learn 2 parameters, strongly nonlinear.") st.image( str(_media / "posterior_ribbons_20_13_k5.png"), caption="Example prior-posterior & truth. L: Difference to true parameter. R: data-sample/output predictived distribution (state-mean [1:3], state-covariance (diag [4:6] and off-diag [7:9]))", width=800, ) st.markdown("- **L96** โ€” Lorenz-96 (40-variable); learn 1-parameter constant forcing.") st.image( str(_media / "posterior_ribbons_const-force_12_1_k3.png"), caption="Example prior-posterior & truth. L: Difference to true parameter. C: parameter-induced forcing of the L96 system. R: data-sample/output predictived distribution ([1:40] state-mean [41:80] state-std)", width=1200, ) st.markdown("- **L96_SPATIAL_FORCING** โ€” Lorenz-96 (40-variable) with spatially-varying forcing; learn 40 parameters; moderately correlated prior.") st.image( str(_media / "posterior_ribbons_vec-force_65_1_k3.png"), caption="Example prior-posterior & truth. L: Difference to true parameters. C: parameter-induced forcing of the L96 system. R: data-sample/output predictived distribution ([1:40] state-mean [41:80] state-std)", width=1200, ) st.markdown("- **L96_NN_FORCING** โ€” Lorenz-96 (100-variable) with a neural-network forcing; Learn 61 parameters (weights and biases of the network). Reasonable prior given.") st.image( str(_media / "posterior_ribbons_flux-force_80_1_k3.png"), caption="Example prior-posterior & truth. L: Difference to true parameters (weights). C: parameter-induced forcing of the L96 system. R: data-sample/output predictived distribution ([1:100] state-mean [101:200] state-std)", width=1200, ) st.subheader("Method taxonomy") st.markdown( "Every method carries four independent tags โ€” how it searches, what update " "mechanism drives each step, what it's built to report, and whether/when it uses " "a surrogate model. See the **๐Ÿ“˜ Methods** page for citations and per-method " "performance charts." ) st.markdown( """ - **Parallelism** โ€” how the search explores parameter space - **Serial** โ€” `ADAM`, `LM` โ€” a single point estimate advanced step by step. - **Parallel-independent** โ€” `ABC`, `HM` โ€” a population of candidates updated with no coupling between members (accepted samples / per-wave resampling). - **Parallel-interacting** โ€” `TEKI`, `ETKI`, `IEKF`, `UKI`, `CES-EKI-DMC`, `CES-EKI-CONST`, `CES-IEKF-CONST` โ€” an ensemble whose members are coupled through a shared update each iteration. - **Update type** โ€” the mechanism driving each update step - **Gradient** โ€” `ADAM`, `LM` โ€” follow the loss gradient (or a Gauss-Newton approximation of it) directly. - **Kalman** โ€” `TEKI`, `ETKI`, `IEKF`, `UKI`, `CES-EKI-DMC`, `CES-EKI-CONST`, `CES-IEKF-CONST` โ€” a (possibly linearized or unscented) Kalman-style ensemble update. - **General** โ€” `ABC`, `HM` โ€” neither gradient- nor Kalman-based (rejection sampling, implausibility cuts). - **Method goal** โ€” what the method is built to report - **Optimization** โ€” `TEKI`, `ETKI`, `UKI`, `ADAM`, `LM` โ€” a single best-fit parameter estimate. - **UQ** โ€” `IEKF`, `ABC`, `HM`, `CES-EKI-DMC`, `CES-EKI-CONST`, `CES-IEKF-CONST` โ€” the full posterior / parameter uncertainty. Can still be scored on the Optimization leaderboard, but tends to be less competitive there since it's not optimizing for speed-to-target. - **Emulator use** โ€” when/whether a surrogate model of the forward model is used - **None** โ€” `TEKI`, `ETKI`, `IEKF`, `UKI`, `ADAM`, `LM`, `ABC` โ€” samples/evaluates the true forward model throughout. - **Within-optimize** โ€” `HM` โ€” refits a surrogate at each iteration (wave) of the search itself. - **After-optimize** โ€” `CES-EKI-DMC`, `CES-EKI-CONST`, `CES-IEKF-CONST` โ€” fits a surrogate (e.g. a GP) once, after calibration finishes, and samples the posterior through it. """ ) st.caption( "Note: Kalman methods are Bayesian in spirit too (they're approximate Gaussian " "posterior updates) โ€” update type is about mechanism (gradient vs. Kalman vs. " "general), not whether a method is 'Bayesian'." ) st.subheader("Key metric") st.markdown( "The reported metric is the **mean number of forward-model evaluations** required " "to reach the target, averaged over random seeds. " "A value of **โˆ’1** indicates a failed run (did not reach the target); " "the **failure rate** shows the fraction of seeds that failed." )