Np7Y3AEVNq / data /claim5_minimax.json
Dinesh Jinjala
Sync data/* runtimes to canonical HF run c7cd5e5b
eecbb57
Raw
History Blame Contribute Delete
12.8 kB
{
"contract": "Theorem 4.9 + Cor 4.10: for the linear class on the hard instance, inf_policy inf_estimator sup_theta E[R(tau_hat)] >= c1*sqrt(d/B); the orthogonalized estimator (Algorithm 2) achieves this rate up to logarithmic factors.",
"proof_reconstruction_ok": true,
"proof_detail": {
"kl_bound_eq49": {
"holds": true,
"c_kl_used": 5.333333333333333,
"tightest_c_kl": 2.197224577336219,
"max_kl": 0.5493061443340548,
"sample": {
"delta": 0.5,
"kl": 0.5493061443340548,
"bound": 1.3333333333333333
}
},
"pinsker": {
"holds": true,
"worst_slack": 0.0
},
"chain_rule_note": "KL(P_th||P_th') = sum_t E[KL(cond_t)] for a fixed policy (standard product-of-conditionals chain rule); one-step KL bounded by Eq.49; sum_j E[N_j]=B."
},
"delta_choice": {
"delta": 0.015309310892394864,
"delta_sq": 0.00023437500000000005,
"is_theta_d_over_B": true,
"lower_bound_const_c1": 0.004687500000000001,
"rate_value": 0.0006629126073623884
},
"slopes": {
"algorithm1/ols": -0.459,
"algorithm1/oracle_constant": -0.0,
"algorithm1/ridge": -0.435,
"algorithm1/segment_mean": -0.459,
"random/ols": -0.51,
"random/oracle_constant": -0.0,
"random/ridge": -0.482,
"random/segment_mean": -0.51,
"uncertainty/ols": -0.459,
"uncertainty/oracle_constant": -0.0,
"uncertainty/ridge": -0.435,
"uncertainty/segment_mean": -0.459
},
"efficient_estimator_slope": -0.459,
"no_estimator_beats_rate": true,
"estimator_matches_rate": true,
"control_hard_over_easy_pehe_ratio": 2.308,
"control_expected_sqrtd": 2.0,
"control_shows_sqrtd_difficulty": true,
"raw_rows": [
{
"policy": "random",
"estimator": "ols",
"B": 50,
"mean_pehe": 0.3702614125112976,
"std_pehe": 0.09873002394721776
},
{
"policy": "random",
"estimator": "ols",
"B": 100,
"mean_pehe": 0.2658632567817004,
"std_pehe": 0.09520651564720185
},
{
"policy": "random",
"estimator": "ols",
"B": 200,
"mean_pehe": 0.17916539750902735,
"std_pehe": 0.06328237544570993
},
{
"policy": "random",
"estimator": "ols",
"B": 400,
"mean_pehe": 0.13324082422338485,
"std_pehe": 0.04511717033648055
},
{
"policy": "random",
"estimator": "ols",
"B": 800,
"mean_pehe": 0.08926453688624392,
"std_pehe": 0.0315987558005107
},
{
"policy": "random",
"estimator": "ridge",
"B": 50,
"mean_pehe": 0.33968816987692957,
"std_pehe": 0.08535037919058892
},
{
"policy": "random",
"estimator": "ridge",
"B": 100,
"mean_pehe": 0.2551358272815906,
"std_pehe": 0.0908432637971804
},
{
"policy": "random",
"estimator": "ridge",
"B": 200,
"mean_pehe": 0.17562118589337325,
"std_pehe": 0.06206084138607062
},
{
"policy": "random",
"estimator": "ridge",
"B": 400,
"mean_pehe": 0.1319683797495677,
"std_pehe": 0.044691588804074985
},
{
"policy": "random",
"estimator": "ridge",
"B": 800,
"mean_pehe": 0.0889283872459233,
"std_pehe": 0.03146246133688038
},
{
"policy": "random",
"estimator": "segment_mean",
"B": 50,
"mean_pehe": 0.3702614125112976,
"std_pehe": 0.09873002394721776
},
{
"policy": "random",
"estimator": "segment_mean",
"B": 100,
"mean_pehe": 0.26586325678170036,
"std_pehe": 0.09520651564720187
},
{
"policy": "random",
"estimator": "segment_mean",
"B": 200,
"mean_pehe": 0.17916539750902735,
"std_pehe": 0.06328237544570993
},
{
"policy": "random",
"estimator": "segment_mean",
"B": 400,
"mean_pehe": 0.13324082422338485,
"std_pehe": 0.04511717033648055
},
{
"policy": "random",
"estimator": "segment_mean",
"B": 800,
"mean_pehe": 0.08926453688624392,
"std_pehe": 0.0315987558005107
},
{
"policy": "random",
"estimator": "oracle_constant",
"B": 50,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "random",
"estimator": "oracle_constant",
"B": 100,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "random",
"estimator": "oracle_constant",
"B": 200,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "random",
"estimator": "oracle_constant",
"B": 400,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "random",
"estimator": "oracle_constant",
"B": 800,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "uncertainty",
"estimator": "ols",
"B": 50,
"mean_pehe": 0.3530777622160817,
"std_pehe": 0.11000627175092942
},
{
"policy": "uncertainty",
"estimator": "ols",
"B": 100,
"mean_pehe": 0.25762326613856185,
"std_pehe": 0.10364323137941223
},
{
"policy": "uncertainty",
"estimator": "ols",
"B": 200,
"mean_pehe": 0.18421952440705744,
"std_pehe": 0.05769026631276346
},
{
"policy": "uncertainty",
"estimator": "ols",
"B": 400,
"mean_pehe": 0.13251481881167532,
"std_pehe": 0.04640211340706573
},
{
"policy": "uncertainty",
"estimator": "ols",
"B": 800,
"mean_pehe": 0.10032899826139048,
"std_pehe": 0.03858335704937093
},
{
"policy": "uncertainty",
"estimator": "ridge",
"B": 50,
"mean_pehe": 0.3278363037308709,
"std_pehe": 0.10084119821228468
},
{
"policy": "uncertainty",
"estimator": "ridge",
"B": 100,
"mean_pehe": 0.24829682518341217,
"std_pehe": 0.10007057773835099
},
{
"policy": "uncertainty",
"estimator": "ridge",
"B": 200,
"mean_pehe": 0.1811424941808721,
"std_pehe": 0.056735977971622
},
{
"policy": "uncertainty",
"estimator": "ridge",
"B": 400,
"mean_pehe": 0.13142520632323024,
"std_pehe": 0.045961095850012804
},
{
"policy": "uncertainty",
"estimator": "ridge",
"B": 800,
"mean_pehe": 0.09989041777490335,
"std_pehe": 0.03844999358839259
},
{
"policy": "uncertainty",
"estimator": "segment_mean",
"B": 50,
"mean_pehe": 0.3530777622160817,
"std_pehe": 0.11000627175092942
},
{
"policy": "uncertainty",
"estimator": "segment_mean",
"B": 100,
"mean_pehe": 0.25762326613856185,
"std_pehe": 0.10364323137941223
},
{
"policy": "uncertainty",
"estimator": "segment_mean",
"B": 200,
"mean_pehe": 0.18421952440705744,
"std_pehe": 0.05769026631276346
},
{
"policy": "uncertainty",
"estimator": "segment_mean",
"B": 400,
"mean_pehe": 0.13251481881167532,
"std_pehe": 0.04640211340706573
},
{
"policy": "uncertainty",
"estimator": "segment_mean",
"B": 800,
"mean_pehe": 0.10032899826139048,
"std_pehe": 0.03858335704937094
},
{
"policy": "uncertainty",
"estimator": "oracle_constant",
"B": 50,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "uncertainty",
"estimator": "oracle_constant",
"B": 100,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "uncertainty",
"estimator": "oracle_constant",
"B": 200,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "uncertainty",
"estimator": "oracle_constant",
"B": 400,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "uncertainty",
"estimator": "oracle_constant",
"B": 800,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "algorithm1",
"estimator": "ols",
"B": 50,
"mean_pehe": 0.3530777622160817,
"std_pehe": 0.11000627175092942
},
{
"policy": "algorithm1",
"estimator": "ols",
"B": 100,
"mean_pehe": 0.25762326613856185,
"std_pehe": 0.10364323137941223
},
{
"policy": "algorithm1",
"estimator": "ols",
"B": 200,
"mean_pehe": 0.18421952440705744,
"std_pehe": 0.05769026631276346
},
{
"policy": "algorithm1",
"estimator": "ols",
"B": 400,
"mean_pehe": 0.13251481881167532,
"std_pehe": 0.04640211340706573
},
{
"policy": "algorithm1",
"estimator": "ols",
"B": 800,
"mean_pehe": 0.10032899826139048,
"std_pehe": 0.03858335704937093
},
{
"policy": "algorithm1",
"estimator": "ridge",
"B": 50,
"mean_pehe": 0.3278363037308709,
"std_pehe": 0.10084119821228468
},
{
"policy": "algorithm1",
"estimator": "ridge",
"B": 100,
"mean_pehe": 0.24829682518341217,
"std_pehe": 0.10007057773835099
},
{
"policy": "algorithm1",
"estimator": "ridge",
"B": 200,
"mean_pehe": 0.1811424941808721,
"std_pehe": 0.056735977971622
},
{
"policy": "algorithm1",
"estimator": "ridge",
"B": 400,
"mean_pehe": 0.13142520632323024,
"std_pehe": 0.045961095850012804
},
{
"policy": "algorithm1",
"estimator": "ridge",
"B": 800,
"mean_pehe": 0.09989041777490335,
"std_pehe": 0.03844999358839259
},
{
"policy": "algorithm1",
"estimator": "segment_mean",
"B": 50,
"mean_pehe": 0.3530777622160817,
"std_pehe": 0.11000627175092942
},
{
"policy": "algorithm1",
"estimator": "segment_mean",
"B": 100,
"mean_pehe": 0.25762326613856185,
"std_pehe": 0.10364323137941223
},
{
"policy": "algorithm1",
"estimator": "segment_mean",
"B": 200,
"mean_pehe": 0.18421952440705744,
"std_pehe": 0.05769026631276346
},
{
"policy": "algorithm1",
"estimator": "segment_mean",
"B": 400,
"mean_pehe": 0.13251481881167532,
"std_pehe": 0.04640211340706573
},
{
"policy": "algorithm1",
"estimator": "segment_mean",
"B": 800,
"mean_pehe": 0.10032899826139048,
"std_pehe": 0.03858335704937094
},
{
"policy": "algorithm1",
"estimator": "oracle_constant",
"B": 50,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "algorithm1",
"estimator": "oracle_constant",
"B": 100,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "algorithm1",
"estimator": "oracle_constant",
"B": 200,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "algorithm1",
"estimator": "oracle_constant",
"B": 400,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
},
{
"policy": "algorithm1",
"estimator": "oracle_constant",
"B": 800,
"mean_pehe": 0.2795084971874738,
"std_pehe": 5.551115123125783e-17
}
],
"d": 4,
"Bs": [
50,
100,
200,
400,
800
],
"trials": 60,
"verdict": "VERIFIED",
"control": "Difficulty control: hard family (d=4 independent params) vs easy family (1 shared param) at B=200 -- PEHE ratio hard/easy = 2.31 ~= sqrt(d)=2.00, so the sqrt(d) factor genuinely comes from d independent degrees of freedom (both share the 1/sqrt(B) slope; the constant differs by sqrt(d)).",
"notes": "The lower bound is a theorem over ALL policies/estimators, so it is certified by the reconstructed proof (machine-checked KL bound Eq.49, Pinsker, KL chain rule, and the Delta-choice giving c1*sqrt(d/B)). The empirical sweep CORROBORATES that no tested estimator beats the rate; it does not, by itself, constitute a proof over every estimator.",
"runtime_s": 4.0,
"config": {
"seed": 0,
"claim5_trials": 60,
"claim5_d": 4,
"claim5_Bs": [
50,
100,
200,
400,
800
]
},
"run_id": "c7cd5e5b-cc0e-4f03-be60-8e43effc2ac3",
"compute": "Hugging Face cpu-upgrade (CPU-only)"
}