ProCreations's picture
Publish validated ICML reproduction
4adaccc verified
Raw
History Blame Contribute Delete
10.3 kB
{
"all_gates_pass": true,
"claim_results": [
{
"assessment": "verified",
"magnitude_direction_spurious_all_varied": true,
"max_absolute_residual": 0.0009918842791635835,
"rows": [
{
"absolute_residual": 7.718664566783748e-07,
"b_mean": 0.003,
"b_variance": 4.074074074074074e-05,
"c_mean": 0.0025000000000000005,
"exact_risk": 4.896887428406236e-05,
"lambda": 0.006,
"leading_risk": 4.974074074074073e-05,
"rho": 0.0,
"scale": 0.01
},
{
"absolute_residual": 6.054583435509127e-06,
"b_mean": 0.006,
"b_variance": 0.00016296296296296295,
"c_mean": 0.005000000000000001,
"exact_risk": 0.0001929083795274538,
"lambda": 0.012,
"leading_risk": 0.00019896296296296293,
"rho": 0.0,
"scale": 0.02
},
{
"absolute_residual": 4.658750999031957e-05,
"b_mean": 0.012,
"b_variance": 0.0006518518518518518,
"c_mean": 0.010000000000000002,
"exact_risk": 0.0007492643418615322,
"lambda": 0.024,
"leading_risk": 0.0007958518518518517,
"rho": 0.0,
"scale": 0.04
},
{
"absolute_residual": 0.0003453137057866456,
"b_mean": 0.024,
"b_variance": 0.0026074074074074072,
"c_mean": 0.020000000000000004,
"exact_risk": 0.0028380937016207614,
"lambda": 0.048,
"leading_risk": 0.003183407407407407,
"rho": 0.0,
"scale": 0.08
},
{
"absolute_residual": 9.961357997115183e-07,
"b_mean": 0.003,
"b_variance": 4.074074074074074e-05,
"c_mean": 0.0025000000000000005,
"exact_risk": 5.377024596667025e-05,
"lambda": 0.006,
"leading_risk": 5.476638176638177e-05,
"rho": 0.35,
"scale": 0.01
},
{
"absolute_residual": 7.810572087759023e-06,
"b_mean": 0.006,
"b_variance": 0.00016296296296296295,
"c_mean": 0.005000000000000001,
"exact_risk": 0.00021125495497776805,
"lambda": 0.012,
"leading_risk": 0.00021906552706552708,
"rho": 0.35,
"scale": 0.02
},
{
"absolute_residual": 6.006213110590038e-05,
"b_mean": 0.012,
"b_variance": 0.0006518518518518518,
"c_mean": 0.010000000000000002,
"exact_risk": 0.0008161999771562079,
"lambda": 0.024,
"leading_risk": 0.0008762621082621083,
"rho": 0.35,
"scale": 0.04
},
{
"absolute_residual": 0.00044500085561553293,
"b_mean": 0.024,
"b_variance": 0.0026074074074074072,
"c_mean": 0.020000000000000004,
"exact_risk": 0.0030600475774329003,
"lambda": 0.048,
"leading_risk": 0.0035050484330484332,
"rho": 0.35,
"scale": 0.08
},
{
"absolute_residual": 2.3010998217227097e-06,
"b_mean": 0.003,
"b_variance": 4.074074074074074e-05,
"c_mean": 0.0025000000000000005,
"exact_risk": 7.377730325668037e-05,
"lambda": 0.006,
"leading_risk": 7.607840307840308e-05,
"rho": 0.65,
"scale": 0.01
},
{
"absolute_residual": 1.7936292524327433e-05,
"b_mean": 0.006,
"b_variance": 0.00016296296296296295,
"c_mean": 0.005000000000000001,
"exact_risk": 0.0002863773197892849,
"lambda": 0.012,
"leading_risk": 0.0003043136123136123,
"rho": 0.65,
"scale": 0.02
},
{
"absolute_residual": 0.0001364217060632394,
"b_mean": 0.012,
"b_variance": 0.0006518518518518518,
"c_mean": 0.010000000000000002,
"exact_risk": 0.0010808327431912098,
"lambda": 0.024,
"leading_risk": 0.0012172544492544493,
"rho": 0.65,
"scale": 0.04
},
{
"absolute_residual": 0.0009918842791635835,
"b_mean": 0.024,
"b_variance": 0.0026074074074074072,
"c_mean": 0.020000000000000004,
"exact_risk": 0.0038771335178542136,
"lambda": 0.048,
"leading_risk": 0.004869017797017797,
"rho": 0.65,
"scale": 0.08
}
]
},
{
"assessment": "falsified",
"literal_reason": "The registered strict-positivity clause fails at the Corollary-2 identity-covariance constant-b boundary, where lambda=b and both exact and displayed optimal risks are zero.",
"nonuniform_positive_optimal_risk": 0.013199999999999996,
"proportionality_cells": [
{
"b": 0.025,
"lambda_star": 0.025,
"ratio": 1.0
},
{
"b": 0.05,
"lambda_star": 0.05,
"ratio": 1.0
},
{
"b": 0.1,
"lambda_star": 0.1,
"ratio": 1.0
},
{
"b": 0.2,
"lambda_star": 0.2,
"ratio": 1.0
},
{
"b": 0.3,
"lambda_star": 0.3,
"ratio": 1.0
}
],
"registered_conjunction_false": true,
"uniform_exact_fixed_point_risk": 0.0,
"uniform_lambda_star": 0.2,
"uniform_leading_optimal_risk": 6.938893903907228e-18,
"uniform_nonzero_b": 0.2
},
{
"absolute_gap": 0.007788688989797343,
"assessment": "verified",
"b": 0.04,
"deterministic_equivalent": 0.32840317565546095,
"empirical_excess_risk_mean": 0.3361918646452583,
"empirical_standard_error": 0.010001793073583705,
"lambda": 0.12,
"n": 80,
"noise": 0.35,
"p": 88,
"runs": 40
},
{
"assessment": "verified",
"high_noise_opposite_direction": true,
"low_noise_same_direction": true,
"rows": [
{
"baseline_lambda": 0.040000001306544186,
"baseline_risk": 0.22706787310005971,
"lambda_shift": 0.0006924101898007698,
"noise": 0.2,
"positive_b_lambda": 0.040692411496344956,
"positive_b_risk": 0.22469074485377682
},
{
"baseline_lambda": 0.49000000726054777,
"baseline_risk": 0.5283407923570076,
"lambda_shift": -0.0023411987511395838,
"noise": 0.7,
"positive_b_lambda": 0.4876588085094082,
"positive_b_risk": 0.5253440032013051
},
{
"baseline_lambda": 0.9999999854707035,
"baseline_risk": 0.6439132703267791,
"lambda_shift": -0.006853146223264384,
"noise": 1.0,
"positive_b_lambda": 0.9931468392474391,
"positive_b_risk": 0.641600545823265
}
]
},
{
"assessment": "verified",
"baseline_curve": [
0.32665160920878983,
0.29023776265729345,
0.2787006694510658,
0.27495091856658077,
0.27461167000441256,
0.27603543024839766,
0.2784639008764378,
0.28150052872899933,
0.28491865477647577,
0.2885799876905502,
0.29239587038070275,
0.29630724999374924
],
"baseline_optimal_lambda": 0.05,
"baseline_optimal_risk": 0.27461167000441256,
"lambda_grid": [
0.01,
0.02,
0.03,
0.04,
0.05,
0.060000000000000005,
0.06999999999999999,
0.08,
0.09,
0.09999999999999999,
0.11,
0.12
],
"n": 80,
"p": 88,
"reinforcing_curve": [
0.3513827776211365,
0.3040007111707276,
0.28523142124036815,
0.2760933518985985,
0.271500405687264,
0.2694512787111104,
0.2689813276125047,
0.2695612414574329,
0.27087304139394297,
0.27271252235267307,
0.2749417002549069,
0.2774635573944301
],
"reinforcing_optimal_lambda": 0.06999999999999999,
"reinforcing_optimal_risk": 0.2689813276125047,
"released_mechanism": "proportional/perforidge.py, five RRM deployments, Sigma=I, paired seeds",
"risk_improvement": 0.00563034239190785,
"runs_per_condition": 16
}
],
"claims": [
{
"claim": 1,
"literal_claim": "In the population setting, Theorem 1 characterizes excess risk as a function of the magnitude and direction of the performative effect together with spurious features (Section 4, Theorem 1)."
},
{
"claim": 2,
"literal_claim": "Corollary 2 shows the optimal regularization parameter in the population regime is proportional to the strength of the performative effect, with optimal risk remaining strictly positive (Section 4, Corollary 2)."
},
{
"claim": 3,
"literal_claim": "Theorem 3 establishes a deterministic equivalent of the performative fixed point for over-parameterized ridge regression when the number of features exceeds the number of samples (Section 5, Theorem 3)."
},
{
"claim": 4,
"literal_claim": "Theorem 4 shows the optimal regularization moves in the same direction as the performative effect on predictive features under low noise, but in the opposite direction under high noise, in the over-parameterized regime (Section 5, Theorem 4)."
},
{
"claim": 5,
"literal_claim": "Numerical experiments in Section 6 confirm that in the over-parameterized setting, performative effects can improve optimally-regularized risk when performativity reinforces existing trends, contrasting with the population-regime degradation (Section 6)."
}
],
"gates": {
"all_assessments_decisive": true,
"claim1_all_components": true,
"claim1_finite": true,
"claim2_control_positive": true,
"claim2_literal_zero": true,
"claim2_proportional": true,
"claim3_finite": true,
"claim3_gap_within_four_se_plus_finite": true,
"claim4_high_noise": true,
"claim4_low_noise": true,
"claim5_lambda_moves_up": true,
"claim5_risk_improves": true
},
"paper_id": "G4ve69pimc"
}