Spaces:
Running
Running
| { | |
| "contract": "Theorem 4.9 + Cor 4.10: for the linear class on the hard instance, inf_policy inf_estimator sup_theta E[R(tau_hat)] >= c1*sqrt(d/B); the orthogonalized estimator (Algorithm 2) achieves this rate up to logarithmic factors.", | |
| "proof_reconstruction_ok": true, | |
| "proof_detail": { | |
| "kl_bound_eq49": { | |
| "holds": true, | |
| "c_kl_used": 5.333333333333333, | |
| "tightest_c_kl": 2.197224577336219, | |
| "max_kl": 0.5493061443340548, | |
| "sample": { | |
| "delta": 0.5, | |
| "kl": 0.5493061443340548, | |
| "bound": 1.3333333333333333 | |
| } | |
| }, | |
| "pinsker": { | |
| "holds": true, | |
| "worst_slack": 0.0 | |
| }, | |
| "chain_rule_note": "KL(P_th||P_th') = sum_t E[KL(cond_t)] for a fixed policy (standard product-of-conditionals chain rule); one-step KL bounded by Eq.49; sum_j E[N_j]=B." | |
| }, | |
| "delta_choice": { | |
| "delta": 0.015309310892394864, | |
| "delta_sq": 0.00023437500000000005, | |
| "is_theta_d_over_B": true, | |
| "lower_bound_const_c1": 0.004687500000000001, | |
| "rate_value": 0.0006629126073623884 | |
| }, | |
| "slopes": { | |
| "algorithm1/ols": -0.459, | |
| "algorithm1/oracle_constant": -0.0, | |
| "algorithm1/ridge": -0.435, | |
| "algorithm1/segment_mean": -0.459, | |
| "random/ols": -0.51, | |
| "random/oracle_constant": -0.0, | |
| "random/ridge": -0.482, | |
| "random/segment_mean": -0.51, | |
| "uncertainty/ols": -0.459, | |
| "uncertainty/oracle_constant": -0.0, | |
| "uncertainty/ridge": -0.435, | |
| "uncertainty/segment_mean": -0.459 | |
| }, | |
| "efficient_estimator_slope": -0.459, | |
| "no_estimator_beats_rate": true, | |
| "estimator_matches_rate": true, | |
| "control_hard_over_easy_pehe_ratio": 2.308, | |
| "control_expected_sqrtd": 2.0, | |
| "control_shows_sqrtd_difficulty": true, | |
| "raw_rows": [ | |
| { | |
| "policy": "random", | |
| "estimator": "ols", | |
| "B": 50, | |
| "mean_pehe": 0.3702614125112976, | |
| "std_pehe": 0.09873002394721776 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "ols", | |
| "B": 100, | |
| "mean_pehe": 0.2658632567817004, | |
| "std_pehe": 0.09520651564720185 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "ols", | |
| "B": 200, | |
| "mean_pehe": 0.17916539750902735, | |
| "std_pehe": 0.06328237544570993 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "ols", | |
| "B": 400, | |
| "mean_pehe": 0.13324082422338485, | |
| "std_pehe": 0.04511717033648055 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "ols", | |
| "B": 800, | |
| "mean_pehe": 0.08926453688624392, | |
| "std_pehe": 0.0315987558005107 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "ridge", | |
| "B": 50, | |
| "mean_pehe": 0.33968816987692957, | |
| "std_pehe": 0.08535037919058892 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "ridge", | |
| "B": 100, | |
| "mean_pehe": 0.2551358272815906, | |
| "std_pehe": 0.0908432637971804 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "ridge", | |
| "B": 200, | |
| "mean_pehe": 0.17562118589337325, | |
| "std_pehe": 0.06206084138607062 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "ridge", | |
| "B": 400, | |
| "mean_pehe": 0.1319683797495677, | |
| "std_pehe": 0.044691588804074985 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "ridge", | |
| "B": 800, | |
| "mean_pehe": 0.0889283872459233, | |
| "std_pehe": 0.03146246133688038 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "segment_mean", | |
| "B": 50, | |
| "mean_pehe": 0.3702614125112976, | |
| "std_pehe": 0.09873002394721776 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "segment_mean", | |
| "B": 100, | |
| "mean_pehe": 0.26586325678170036, | |
| "std_pehe": 0.09520651564720187 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "segment_mean", | |
| "B": 200, | |
| "mean_pehe": 0.17916539750902735, | |
| "std_pehe": 0.06328237544570993 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "segment_mean", | |
| "B": 400, | |
| "mean_pehe": 0.13324082422338485, | |
| "std_pehe": 0.04511717033648055 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "segment_mean", | |
| "B": 800, | |
| "mean_pehe": 0.08926453688624392, | |
| "std_pehe": 0.0315987558005107 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "oracle_constant", | |
| "B": 50, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "oracle_constant", | |
| "B": 100, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "oracle_constant", | |
| "B": 200, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "oracle_constant", | |
| "B": 400, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "random", | |
| "estimator": "oracle_constant", | |
| "B": 800, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "ols", | |
| "B": 50, | |
| "mean_pehe": 0.3530777622160817, | |
| "std_pehe": 0.11000627175092942 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "ols", | |
| "B": 100, | |
| "mean_pehe": 0.25762326613856185, | |
| "std_pehe": 0.10364323137941223 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "ols", | |
| "B": 200, | |
| "mean_pehe": 0.18421952440705744, | |
| "std_pehe": 0.05769026631276346 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "ols", | |
| "B": 400, | |
| "mean_pehe": 0.13251481881167532, | |
| "std_pehe": 0.04640211340706573 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "ols", | |
| "B": 800, | |
| "mean_pehe": 0.10032899826139048, | |
| "std_pehe": 0.03858335704937093 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "ridge", | |
| "B": 50, | |
| "mean_pehe": 0.3278363037308709, | |
| "std_pehe": 0.10084119821228468 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "ridge", | |
| "B": 100, | |
| "mean_pehe": 0.24829682518341217, | |
| "std_pehe": 0.10007057773835099 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "ridge", | |
| "B": 200, | |
| "mean_pehe": 0.1811424941808721, | |
| "std_pehe": 0.056735977971622 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "ridge", | |
| "B": 400, | |
| "mean_pehe": 0.13142520632323024, | |
| "std_pehe": 0.045961095850012804 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "ridge", | |
| "B": 800, | |
| "mean_pehe": 0.09989041777490335, | |
| "std_pehe": 0.03844999358839259 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "segment_mean", | |
| "B": 50, | |
| "mean_pehe": 0.3530777622160817, | |
| "std_pehe": 0.11000627175092942 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "segment_mean", | |
| "B": 100, | |
| "mean_pehe": 0.25762326613856185, | |
| "std_pehe": 0.10364323137941223 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "segment_mean", | |
| "B": 200, | |
| "mean_pehe": 0.18421952440705744, | |
| "std_pehe": 0.05769026631276346 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "segment_mean", | |
| "B": 400, | |
| "mean_pehe": 0.13251481881167532, | |
| "std_pehe": 0.04640211340706573 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "segment_mean", | |
| "B": 800, | |
| "mean_pehe": 0.10032899826139048, | |
| "std_pehe": 0.03858335704937094 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "oracle_constant", | |
| "B": 50, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "oracle_constant", | |
| "B": 100, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "oracle_constant", | |
| "B": 200, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "oracle_constant", | |
| "B": 400, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "uncertainty", | |
| "estimator": "oracle_constant", | |
| "B": 800, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "ols", | |
| "B": 50, | |
| "mean_pehe": 0.3530777622160817, | |
| "std_pehe": 0.11000627175092942 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "ols", | |
| "B": 100, | |
| "mean_pehe": 0.25762326613856185, | |
| "std_pehe": 0.10364323137941223 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "ols", | |
| "B": 200, | |
| "mean_pehe": 0.18421952440705744, | |
| "std_pehe": 0.05769026631276346 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "ols", | |
| "B": 400, | |
| "mean_pehe": 0.13251481881167532, | |
| "std_pehe": 0.04640211340706573 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "ols", | |
| "B": 800, | |
| "mean_pehe": 0.10032899826139048, | |
| "std_pehe": 0.03858335704937093 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "ridge", | |
| "B": 50, | |
| "mean_pehe": 0.3278363037308709, | |
| "std_pehe": 0.10084119821228468 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "ridge", | |
| "B": 100, | |
| "mean_pehe": 0.24829682518341217, | |
| "std_pehe": 0.10007057773835099 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "ridge", | |
| "B": 200, | |
| "mean_pehe": 0.1811424941808721, | |
| "std_pehe": 0.056735977971622 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "ridge", | |
| "B": 400, | |
| "mean_pehe": 0.13142520632323024, | |
| "std_pehe": 0.045961095850012804 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "ridge", | |
| "B": 800, | |
| "mean_pehe": 0.09989041777490335, | |
| "std_pehe": 0.03844999358839259 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "segment_mean", | |
| "B": 50, | |
| "mean_pehe": 0.3530777622160817, | |
| "std_pehe": 0.11000627175092942 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "segment_mean", | |
| "B": 100, | |
| "mean_pehe": 0.25762326613856185, | |
| "std_pehe": 0.10364323137941223 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "segment_mean", | |
| "B": 200, | |
| "mean_pehe": 0.18421952440705744, | |
| "std_pehe": 0.05769026631276346 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "segment_mean", | |
| "B": 400, | |
| "mean_pehe": 0.13251481881167532, | |
| "std_pehe": 0.04640211340706573 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "segment_mean", | |
| "B": 800, | |
| "mean_pehe": 0.10032899826139048, | |
| "std_pehe": 0.03858335704937094 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "oracle_constant", | |
| "B": 50, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "oracle_constant", | |
| "B": 100, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "oracle_constant", | |
| "B": 200, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "oracle_constant", | |
| "B": 400, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| }, | |
| { | |
| "policy": "algorithm1", | |
| "estimator": "oracle_constant", | |
| "B": 800, | |
| "mean_pehe": 0.2795084971874738, | |
| "std_pehe": 5.551115123125783e-17 | |
| } | |
| ], | |
| "d": 4, | |
| "Bs": [ | |
| 50, | |
| 100, | |
| 200, | |
| 400, | |
| 800 | |
| ], | |
| "trials": 60, | |
| "verdict": "VERIFIED", | |
| "control": "Difficulty control: hard family (d=4 independent params) vs easy family (1 shared param) at B=200 -- PEHE ratio hard/easy = 2.31 ~= sqrt(d)=2.00, so the sqrt(d) factor genuinely comes from d independent degrees of freedom (both share the 1/sqrt(B) slope; the constant differs by sqrt(d)).", | |
| "notes": "The lower bound is a theorem over ALL policies/estimators, so it is certified by the reconstructed proof (machine-checked KL bound Eq.49, Pinsker, KL chain rule, and the Delta-choice giving c1*sqrt(d/B)). The empirical sweep CORROBORATES that no tested estimator beats the rate; it does not, by itself, constitute a proof over every estimator.", | |
| "runtime_s": 4.0, | |
| "config": { | |
| "seed": 0, | |
| "claim5_trials": 60, | |
| "claim5_d": 4, | |
| "claim5_Bs": [ | |
| 50, | |
| 100, | |
| 200, | |
| 400, | |
| 800 | |
| ] | |
| }, | |
| "run_id": "c7cd5e5b-cc0e-4f03-be60-8e43effc2ac3", | |
| "compute": "Hugging Face cpu-upgrade (CPU-only)" | |
| } |