{ "contract": "Theorem 4.9 + Cor 4.10: for the linear class on the hard instance, inf_policy inf_estimator sup_theta E[R(tau_hat)] >= c1*sqrt(d/B); the orthogonalized estimator (Algorithm 2) achieves this rate up to logarithmic factors.", "proof_reconstruction_ok": true, "proof_detail": { "kl_bound_eq49": { "holds": true, "c_kl_used": 5.333333333333333, "tightest_c_kl": 2.197224577336219, "max_kl": 0.5493061443340548, "sample": { "delta": 0.5, "kl": 0.5493061443340548, "bound": 1.3333333333333333 } }, "pinsker": { "holds": true, "worst_slack": 0.0 }, "chain_rule_note": "KL(P_th||P_th') = sum_t E[KL(cond_t)] for a fixed policy (standard product-of-conditionals chain rule); one-step KL bounded by Eq.49; sum_j E[N_j]=B." }, "delta_choice": { "delta": 0.015309310892394864, "delta_sq": 0.00023437500000000005, "is_theta_d_over_B": true, "lower_bound_const_c1": 0.004687500000000001, "rate_value": 0.0006629126073623884 }, "slopes": { "algorithm1/ols": -0.459, "algorithm1/oracle_constant": -0.0, "algorithm1/ridge": -0.435, "algorithm1/segment_mean": -0.459, "random/ols": -0.51, "random/oracle_constant": -0.0, "random/ridge": -0.482, "random/segment_mean": -0.51, "uncertainty/ols": -0.459, "uncertainty/oracle_constant": -0.0, "uncertainty/ridge": -0.435, "uncertainty/segment_mean": -0.459 }, "efficient_estimator_slope": -0.459, "no_estimator_beats_rate": true, "estimator_matches_rate": true, "control_hard_over_easy_pehe_ratio": 2.308, "control_expected_sqrtd": 2.0, "control_shows_sqrtd_difficulty": true, "raw_rows": [ { "policy": "random", "estimator": "ols", "B": 50, "mean_pehe": 0.3702614125112976, "std_pehe": 0.09873002394721776 }, { "policy": "random", "estimator": "ols", "B": 100, "mean_pehe": 0.2658632567817004, "std_pehe": 0.09520651564720185 }, { "policy": "random", "estimator": "ols", "B": 200, "mean_pehe": 0.17916539750902735, "std_pehe": 0.06328237544570993 }, { "policy": "random", "estimator": "ols", "B": 400, "mean_pehe": 0.13324082422338485, "std_pehe": 0.04511717033648055 }, { "policy": "random", "estimator": "ols", "B": 800, "mean_pehe": 0.08926453688624392, "std_pehe": 0.0315987558005107 }, { "policy": "random", "estimator": "ridge", "B": 50, "mean_pehe": 0.33968816987692957, "std_pehe": 0.08535037919058892 }, { "policy": "random", "estimator": "ridge", "B": 100, "mean_pehe": 0.2551358272815906, "std_pehe": 0.0908432637971804 }, { "policy": "random", "estimator": "ridge", "B": 200, "mean_pehe": 0.17562118589337325, "std_pehe": 0.06206084138607062 }, { "policy": "random", "estimator": "ridge", "B": 400, "mean_pehe": 0.1319683797495677, "std_pehe": 0.044691588804074985 }, { "policy": "random", "estimator": "ridge", "B": 800, "mean_pehe": 0.0889283872459233, "std_pehe": 0.03146246133688038 }, { "policy": "random", "estimator": "segment_mean", "B": 50, "mean_pehe": 0.3702614125112976, "std_pehe": 0.09873002394721776 }, { "policy": "random", "estimator": "segment_mean", "B": 100, "mean_pehe": 0.26586325678170036, "std_pehe": 0.09520651564720187 }, { "policy": "random", "estimator": "segment_mean", "B": 200, "mean_pehe": 0.17916539750902735, "std_pehe": 0.06328237544570993 }, { "policy": "random", "estimator": "segment_mean", "B": 400, "mean_pehe": 0.13324082422338485, "std_pehe": 0.04511717033648055 }, { "policy": "random", "estimator": "segment_mean", "B": 800, "mean_pehe": 0.08926453688624392, "std_pehe": 0.0315987558005107 }, { "policy": "random", "estimator": "oracle_constant", "B": 50, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "random", "estimator": "oracle_constant", "B": 100, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "random", "estimator": "oracle_constant", "B": 200, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "random", "estimator": "oracle_constant", "B": 400, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "random", "estimator": "oracle_constant", "B": 800, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "uncertainty", "estimator": "ols", "B": 50, "mean_pehe": 0.3530777622160817, "std_pehe": 0.11000627175092942 }, { "policy": "uncertainty", "estimator": "ols", "B": 100, "mean_pehe": 0.25762326613856185, "std_pehe": 0.10364323137941223 }, { "policy": "uncertainty", "estimator": "ols", "B": 200, "mean_pehe": 0.18421952440705744, "std_pehe": 0.05769026631276346 }, { "policy": "uncertainty", "estimator": "ols", "B": 400, "mean_pehe": 0.13251481881167532, "std_pehe": 0.04640211340706573 }, { "policy": "uncertainty", "estimator": "ols", "B": 800, "mean_pehe": 0.10032899826139048, "std_pehe": 0.03858335704937093 }, { "policy": "uncertainty", "estimator": "ridge", "B": 50, "mean_pehe": 0.3278363037308709, "std_pehe": 0.10084119821228468 }, { "policy": "uncertainty", "estimator": "ridge", "B": 100, "mean_pehe": 0.24829682518341217, "std_pehe": 0.10007057773835099 }, { "policy": "uncertainty", "estimator": "ridge", "B": 200, "mean_pehe": 0.1811424941808721, "std_pehe": 0.056735977971622 }, { "policy": "uncertainty", "estimator": "ridge", "B": 400, "mean_pehe": 0.13142520632323024, "std_pehe": 0.045961095850012804 }, { "policy": "uncertainty", "estimator": "ridge", "B": 800, "mean_pehe": 0.09989041777490335, "std_pehe": 0.03844999358839259 }, { "policy": "uncertainty", "estimator": "segment_mean", "B": 50, "mean_pehe": 0.3530777622160817, "std_pehe": 0.11000627175092942 }, { "policy": "uncertainty", "estimator": "segment_mean", "B": 100, "mean_pehe": 0.25762326613856185, "std_pehe": 0.10364323137941223 }, { "policy": "uncertainty", "estimator": "segment_mean", "B": 200, "mean_pehe": 0.18421952440705744, "std_pehe": 0.05769026631276346 }, { "policy": "uncertainty", "estimator": "segment_mean", "B": 400, "mean_pehe": 0.13251481881167532, "std_pehe": 0.04640211340706573 }, { "policy": "uncertainty", "estimator": "segment_mean", "B": 800, "mean_pehe": 0.10032899826139048, "std_pehe": 0.03858335704937094 }, { "policy": "uncertainty", "estimator": "oracle_constant", "B": 50, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "uncertainty", "estimator": "oracle_constant", "B": 100, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "uncertainty", "estimator": "oracle_constant", "B": 200, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "uncertainty", "estimator": "oracle_constant", "B": 400, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "uncertainty", "estimator": "oracle_constant", "B": 800, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "algorithm1", "estimator": "ols", "B": 50, "mean_pehe": 0.3530777622160817, "std_pehe": 0.11000627175092942 }, { "policy": "algorithm1", "estimator": "ols", "B": 100, "mean_pehe": 0.25762326613856185, "std_pehe": 0.10364323137941223 }, { "policy": "algorithm1", "estimator": "ols", "B": 200, "mean_pehe": 0.18421952440705744, "std_pehe": 0.05769026631276346 }, { "policy": "algorithm1", "estimator": "ols", "B": 400, "mean_pehe": 0.13251481881167532, "std_pehe": 0.04640211340706573 }, { "policy": "algorithm1", "estimator": "ols", "B": 800, "mean_pehe": 0.10032899826139048, "std_pehe": 0.03858335704937093 }, { "policy": "algorithm1", "estimator": "ridge", "B": 50, "mean_pehe": 0.3278363037308709, "std_pehe": 0.10084119821228468 }, { "policy": "algorithm1", "estimator": "ridge", "B": 100, "mean_pehe": 0.24829682518341217, "std_pehe": 0.10007057773835099 }, { "policy": "algorithm1", "estimator": "ridge", "B": 200, "mean_pehe": 0.1811424941808721, "std_pehe": 0.056735977971622 }, { "policy": "algorithm1", "estimator": "ridge", "B": 400, "mean_pehe": 0.13142520632323024, "std_pehe": 0.045961095850012804 }, { "policy": "algorithm1", "estimator": "ridge", "B": 800, "mean_pehe": 0.09989041777490335, "std_pehe": 0.03844999358839259 }, { "policy": "algorithm1", "estimator": "segment_mean", "B": 50, "mean_pehe": 0.3530777622160817, "std_pehe": 0.11000627175092942 }, { "policy": "algorithm1", "estimator": "segment_mean", "B": 100, "mean_pehe": 0.25762326613856185, "std_pehe": 0.10364323137941223 }, { "policy": "algorithm1", "estimator": "segment_mean", "B": 200, "mean_pehe": 0.18421952440705744, "std_pehe": 0.05769026631276346 }, { "policy": "algorithm1", "estimator": "segment_mean", "B": 400, "mean_pehe": 0.13251481881167532, "std_pehe": 0.04640211340706573 }, { "policy": "algorithm1", "estimator": "segment_mean", "B": 800, "mean_pehe": 0.10032899826139048, "std_pehe": 0.03858335704937094 }, { "policy": "algorithm1", "estimator": "oracle_constant", "B": 50, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "algorithm1", "estimator": "oracle_constant", "B": 100, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "algorithm1", "estimator": "oracle_constant", "B": 200, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "algorithm1", "estimator": "oracle_constant", "B": 400, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 }, { "policy": "algorithm1", "estimator": "oracle_constant", "B": 800, "mean_pehe": 0.2795084971874738, "std_pehe": 5.551115123125783e-17 } ], "d": 4, "Bs": [ 50, 100, 200, 400, 800 ], "trials": 60, "verdict": "VERIFIED", "control": "Difficulty control: hard family (d=4 independent params) vs easy family (1 shared param) at B=200 -- PEHE ratio hard/easy = 2.31 ~= sqrt(d)=2.00, so the sqrt(d) factor genuinely comes from d independent degrees of freedom (both share the 1/sqrt(B) slope; the constant differs by sqrt(d)).", "notes": "The lower bound is a theorem over ALL policies/estimators, so it is certified by the reconstructed proof (machine-checked KL bound Eq.49, Pinsker, KL chain rule, and the Delta-choice giving c1*sqrt(d/B)). The empirical sweep CORROBORATES that no tested estimator beats the rate; it does not, by itself, constitute a proof over every estimator.", "runtime_s": 4.0, "config": { "seed": 0, "claim5_trials": 60, "claim5_d": 4, "claim5_Bs": [ 50, 100, 200, 400, 800 ] }, "run_id": "c7cd5e5b-cc0e-4f03-be60-8e43effc2ac3", "compute": "Hugging Face cpu-upgrade (CPU-only)" }