{ "all_gates_pass": true, "claim_results": [ { "assessment": "verified", "magnitude_direction_spurious_all_varied": true, "max_absolute_residual": 0.0009918842791635835, "rows": [ { "absolute_residual": 7.718664566783748e-07, "b_mean": 0.003, "b_variance": 4.074074074074074e-05, "c_mean": 0.0025000000000000005, "exact_risk": 4.896887428406236e-05, "lambda": 0.006, "leading_risk": 4.974074074074073e-05, "rho": 0.0, "scale": 0.01 }, { "absolute_residual": 6.054583435509127e-06, "b_mean": 0.006, "b_variance": 0.00016296296296296295, "c_mean": 0.005000000000000001, "exact_risk": 0.0001929083795274538, "lambda": 0.012, "leading_risk": 0.00019896296296296293, "rho": 0.0, "scale": 0.02 }, { "absolute_residual": 4.658750999031957e-05, "b_mean": 0.012, "b_variance": 0.0006518518518518518, "c_mean": 0.010000000000000002, "exact_risk": 0.0007492643418615322, "lambda": 0.024, "leading_risk": 0.0007958518518518517, "rho": 0.0, "scale": 0.04 }, { "absolute_residual": 0.0003453137057866456, "b_mean": 0.024, "b_variance": 0.0026074074074074072, "c_mean": 0.020000000000000004, "exact_risk": 0.0028380937016207614, "lambda": 0.048, "leading_risk": 0.003183407407407407, "rho": 0.0, "scale": 0.08 }, { "absolute_residual": 9.961357997115183e-07, "b_mean": 0.003, "b_variance": 4.074074074074074e-05, "c_mean": 0.0025000000000000005, "exact_risk": 5.377024596667025e-05, "lambda": 0.006, "leading_risk": 5.476638176638177e-05, "rho": 0.35, "scale": 0.01 }, { "absolute_residual": 7.810572087759023e-06, "b_mean": 0.006, "b_variance": 0.00016296296296296295, "c_mean": 0.005000000000000001, "exact_risk": 0.00021125495497776805, "lambda": 0.012, "leading_risk": 0.00021906552706552708, "rho": 0.35, "scale": 0.02 }, { "absolute_residual": 6.006213110590038e-05, "b_mean": 0.012, "b_variance": 0.0006518518518518518, "c_mean": 0.010000000000000002, "exact_risk": 0.0008161999771562079, "lambda": 0.024, "leading_risk": 0.0008762621082621083, "rho": 0.35, "scale": 0.04 }, { "absolute_residual": 0.00044500085561553293, "b_mean": 0.024, "b_variance": 0.0026074074074074072, "c_mean": 0.020000000000000004, "exact_risk": 0.0030600475774329003, "lambda": 0.048, "leading_risk": 0.0035050484330484332, "rho": 0.35, "scale": 0.08 }, { "absolute_residual": 2.3010998217227097e-06, "b_mean": 0.003, "b_variance": 4.074074074074074e-05, "c_mean": 0.0025000000000000005, "exact_risk": 7.377730325668037e-05, "lambda": 0.006, "leading_risk": 7.607840307840308e-05, "rho": 0.65, "scale": 0.01 }, { "absolute_residual": 1.7936292524327433e-05, "b_mean": 0.006, "b_variance": 0.00016296296296296295, "c_mean": 0.005000000000000001, "exact_risk": 0.0002863773197892849, "lambda": 0.012, "leading_risk": 0.0003043136123136123, "rho": 0.65, "scale": 0.02 }, { "absolute_residual": 0.0001364217060632394, "b_mean": 0.012, "b_variance": 0.0006518518518518518, "c_mean": 0.010000000000000002, "exact_risk": 0.0010808327431912098, "lambda": 0.024, "leading_risk": 0.0012172544492544493, "rho": 0.65, "scale": 0.04 }, { "absolute_residual": 0.0009918842791635835, "b_mean": 0.024, "b_variance": 0.0026074074074074072, "c_mean": 0.020000000000000004, "exact_risk": 0.0038771335178542136, "lambda": 0.048, "leading_risk": 0.004869017797017797, "rho": 0.65, "scale": 0.08 } ] }, { "assessment": "falsified", "literal_reason": "The registered strict-positivity clause fails at the Corollary-2 identity-covariance constant-b boundary, where lambda=b and both exact and displayed optimal risks are zero.", "nonuniform_positive_optimal_risk": 0.013199999999999996, "proportionality_cells": [ { "b": 0.025, "lambda_star": 0.025, "ratio": 1.0 }, { "b": 0.05, "lambda_star": 0.05, "ratio": 1.0 }, { "b": 0.1, "lambda_star": 0.1, "ratio": 1.0 }, { "b": 0.2, "lambda_star": 0.2, "ratio": 1.0 }, { "b": 0.3, "lambda_star": 0.3, "ratio": 1.0 } ], "registered_conjunction_false": true, "uniform_exact_fixed_point_risk": 0.0, "uniform_lambda_star": 0.2, "uniform_leading_optimal_risk": 6.938893903907228e-18, "uniform_nonzero_b": 0.2 }, { "absolute_gap": 0.007788688989797343, "assessment": "verified", "b": 0.04, "deterministic_equivalent": 0.32840317565546095, "empirical_excess_risk_mean": 0.3361918646452583, "empirical_standard_error": 0.010001793073583705, "lambda": 0.12, "n": 80, "noise": 0.35, "p": 88, "runs": 40 }, { "assessment": "verified", "high_noise_opposite_direction": true, "low_noise_same_direction": true, "rows": [ { "baseline_lambda": 0.040000001306544186, "baseline_risk": 0.22706787310005971, "lambda_shift": 0.0006924101898007698, "noise": 0.2, "positive_b_lambda": 0.040692411496344956, "positive_b_risk": 0.22469074485377682 }, { "baseline_lambda": 0.49000000726054777, "baseline_risk": 0.5283407923570076, "lambda_shift": -0.0023411987511395838, "noise": 0.7, "positive_b_lambda": 0.4876588085094082, "positive_b_risk": 0.5253440032013051 }, { "baseline_lambda": 0.9999999854707035, "baseline_risk": 0.6439132703267791, "lambda_shift": -0.006853146223264384, "noise": 1.0, "positive_b_lambda": 0.9931468392474391, "positive_b_risk": 0.641600545823265 } ] }, { "assessment": "verified", "baseline_curve": [ 0.32665160920878983, 0.29023776265729345, 0.2787006694510658, 0.27495091856658077, 0.27461167000441256, 0.27603543024839766, 0.2784639008764378, 0.28150052872899933, 0.28491865477647577, 0.2885799876905502, 0.29239587038070275, 0.29630724999374924 ], "baseline_optimal_lambda": 0.05, "baseline_optimal_risk": 0.27461167000441256, "lambda_grid": [ 0.01, 0.02, 0.03, 0.04, 0.05, 0.060000000000000005, 0.06999999999999999, 0.08, 0.09, 0.09999999999999999, 0.11, 0.12 ], "n": 80, "p": 88, "reinforcing_curve": [ 0.3513827776211365, 0.3040007111707276, 0.28523142124036815, 0.2760933518985985, 0.271500405687264, 0.2694512787111104, 0.2689813276125047, 0.2695612414574329, 0.27087304139394297, 0.27271252235267307, 0.2749417002549069, 0.2774635573944301 ], "reinforcing_optimal_lambda": 0.06999999999999999, "reinforcing_optimal_risk": 0.2689813276125047, "released_mechanism": "proportional/perforidge.py, five RRM deployments, Sigma=I, paired seeds", "risk_improvement": 0.00563034239190785, "runs_per_condition": 16 } ], "claims": [ { "claim": 1, "literal_claim": "In the population setting, Theorem 1 characterizes excess risk as a function of the magnitude and direction of the performative effect together with spurious features (Section 4, Theorem 1)." }, { "claim": 2, "literal_claim": "Corollary 2 shows the optimal regularization parameter in the population regime is proportional to the strength of the performative effect, with optimal risk remaining strictly positive (Section 4, Corollary 2)." }, { "claim": 3, "literal_claim": "Theorem 3 establishes a deterministic equivalent of the performative fixed point for over-parameterized ridge regression when the number of features exceeds the number of samples (Section 5, Theorem 3)." }, { "claim": 4, "literal_claim": "Theorem 4 shows the optimal regularization moves in the same direction as the performative effect on predictive features under low noise, but in the opposite direction under high noise, in the over-parameterized regime (Section 5, Theorem 4)." }, { "claim": 5, "literal_claim": "Numerical experiments in Section 6 confirm that in the over-parameterized setting, performative effects can improve optimally-regularized risk when performativity reinforces existing trends, contrasting with the population-regime degradation (Section 6)." } ], "gates": { "all_assessments_decisive": true, "claim1_all_components": true, "claim1_finite": true, "claim2_control_positive": true, "claim2_literal_zero": true, "claim2_proportional": true, "claim3_finite": true, "claim3_gap_within_four_se_plus_finite": true, "claim4_high_noise": true, "claim4_low_noise": true, "claim5_lambda_moves_up": true, "claim5_risk_improves": true }, "paper_id": "G4ve69pimc" }