| { |
| "all_gates_pass": true, |
| "claim_results": [ |
| { |
| "assessment": "verified", |
| "magnitude_direction_spurious_all_varied": true, |
| "max_absolute_residual": 0.0009918842791635835, |
| "rows": [ |
| { |
| "absolute_residual": 7.718664566783748e-07, |
| "b_mean": 0.003, |
| "b_variance": 4.074074074074074e-05, |
| "c_mean": 0.0025000000000000005, |
| "exact_risk": 4.896887428406236e-05, |
| "lambda": 0.006, |
| "leading_risk": 4.974074074074073e-05, |
| "rho": 0.0, |
| "scale": 0.01 |
| }, |
| { |
| "absolute_residual": 6.054583435509127e-06, |
| "b_mean": 0.006, |
| "b_variance": 0.00016296296296296295, |
| "c_mean": 0.005000000000000001, |
| "exact_risk": 0.0001929083795274538, |
| "lambda": 0.012, |
| "leading_risk": 0.00019896296296296293, |
| "rho": 0.0, |
| "scale": 0.02 |
| }, |
| { |
| "absolute_residual": 4.658750999031957e-05, |
| "b_mean": 0.012, |
| "b_variance": 0.0006518518518518518, |
| "c_mean": 0.010000000000000002, |
| "exact_risk": 0.0007492643418615322, |
| "lambda": 0.024, |
| "leading_risk": 0.0007958518518518517, |
| "rho": 0.0, |
| "scale": 0.04 |
| }, |
| { |
| "absolute_residual": 0.0003453137057866456, |
| "b_mean": 0.024, |
| "b_variance": 0.0026074074074074072, |
| "c_mean": 0.020000000000000004, |
| "exact_risk": 0.0028380937016207614, |
| "lambda": 0.048, |
| "leading_risk": 0.003183407407407407, |
| "rho": 0.0, |
| "scale": 0.08 |
| }, |
| { |
| "absolute_residual": 9.961357997115183e-07, |
| "b_mean": 0.003, |
| "b_variance": 4.074074074074074e-05, |
| "c_mean": 0.0025000000000000005, |
| "exact_risk": 5.377024596667025e-05, |
| "lambda": 0.006, |
| "leading_risk": 5.476638176638177e-05, |
| "rho": 0.35, |
| "scale": 0.01 |
| }, |
| { |
| "absolute_residual": 7.810572087759023e-06, |
| "b_mean": 0.006, |
| "b_variance": 0.00016296296296296295, |
| "c_mean": 0.005000000000000001, |
| "exact_risk": 0.00021125495497776805, |
| "lambda": 0.012, |
| "leading_risk": 0.00021906552706552708, |
| "rho": 0.35, |
| "scale": 0.02 |
| }, |
| { |
| "absolute_residual": 6.006213110590038e-05, |
| "b_mean": 0.012, |
| "b_variance": 0.0006518518518518518, |
| "c_mean": 0.010000000000000002, |
| "exact_risk": 0.0008161999771562079, |
| "lambda": 0.024, |
| "leading_risk": 0.0008762621082621083, |
| "rho": 0.35, |
| "scale": 0.04 |
| }, |
| { |
| "absolute_residual": 0.00044500085561553293, |
| "b_mean": 0.024, |
| "b_variance": 0.0026074074074074072, |
| "c_mean": 0.020000000000000004, |
| "exact_risk": 0.0030600475774329003, |
| "lambda": 0.048, |
| "leading_risk": 0.0035050484330484332, |
| "rho": 0.35, |
| "scale": 0.08 |
| }, |
| { |
| "absolute_residual": 2.3010998217227097e-06, |
| "b_mean": 0.003, |
| "b_variance": 4.074074074074074e-05, |
| "c_mean": 0.0025000000000000005, |
| "exact_risk": 7.377730325668037e-05, |
| "lambda": 0.006, |
| "leading_risk": 7.607840307840308e-05, |
| "rho": 0.65, |
| "scale": 0.01 |
| }, |
| { |
| "absolute_residual": 1.7936292524327433e-05, |
| "b_mean": 0.006, |
| "b_variance": 0.00016296296296296295, |
| "c_mean": 0.005000000000000001, |
| "exact_risk": 0.0002863773197892849, |
| "lambda": 0.012, |
| "leading_risk": 0.0003043136123136123, |
| "rho": 0.65, |
| "scale": 0.02 |
| }, |
| { |
| "absolute_residual": 0.0001364217060632394, |
| "b_mean": 0.012, |
| "b_variance": 0.0006518518518518518, |
| "c_mean": 0.010000000000000002, |
| "exact_risk": 0.0010808327431912098, |
| "lambda": 0.024, |
| "leading_risk": 0.0012172544492544493, |
| "rho": 0.65, |
| "scale": 0.04 |
| }, |
| { |
| "absolute_residual": 0.0009918842791635835, |
| "b_mean": 0.024, |
| "b_variance": 0.0026074074074074072, |
| "c_mean": 0.020000000000000004, |
| "exact_risk": 0.0038771335178542136, |
| "lambda": 0.048, |
| "leading_risk": 0.004869017797017797, |
| "rho": 0.65, |
| "scale": 0.08 |
| } |
| ] |
| }, |
| { |
| "assessment": "falsified", |
| "literal_reason": "The registered strict-positivity clause fails at the Corollary-2 identity-covariance constant-b boundary, where lambda=b and both exact and displayed optimal risks are zero.", |
| "nonuniform_positive_optimal_risk": 0.013199999999999996, |
| "proportionality_cells": [ |
| { |
| "b": 0.025, |
| "lambda_star": 0.025, |
| "ratio": 1.0 |
| }, |
| { |
| "b": 0.05, |
| "lambda_star": 0.05, |
| "ratio": 1.0 |
| }, |
| { |
| "b": 0.1, |
| "lambda_star": 0.1, |
| "ratio": 1.0 |
| }, |
| { |
| "b": 0.2, |
| "lambda_star": 0.2, |
| "ratio": 1.0 |
| }, |
| { |
| "b": 0.3, |
| "lambda_star": 0.3, |
| "ratio": 1.0 |
| } |
| ], |
| "registered_conjunction_false": true, |
| "uniform_exact_fixed_point_risk": 0.0, |
| "uniform_lambda_star": 0.2, |
| "uniform_leading_optimal_risk": 6.938893903907228e-18, |
| "uniform_nonzero_b": 0.2 |
| }, |
| { |
| "absolute_gap": 0.007788688989797343, |
| "assessment": "verified", |
| "b": 0.04, |
| "deterministic_equivalent": 0.32840317565546095, |
| "empirical_excess_risk_mean": 0.3361918646452583, |
| "empirical_standard_error": 0.010001793073583705, |
| "lambda": 0.12, |
| "n": 80, |
| "noise": 0.35, |
| "p": 88, |
| "runs": 40 |
| }, |
| { |
| "assessment": "verified", |
| "high_noise_opposite_direction": true, |
| "low_noise_same_direction": true, |
| "rows": [ |
| { |
| "baseline_lambda": 0.040000001306544186, |
| "baseline_risk": 0.22706787310005971, |
| "lambda_shift": 0.0006924101898007698, |
| "noise": 0.2, |
| "positive_b_lambda": 0.040692411496344956, |
| "positive_b_risk": 0.22469074485377682 |
| }, |
| { |
| "baseline_lambda": 0.49000000726054777, |
| "baseline_risk": 0.5283407923570076, |
| "lambda_shift": -0.0023411987511395838, |
| "noise": 0.7, |
| "positive_b_lambda": 0.4876588085094082, |
| "positive_b_risk": 0.5253440032013051 |
| }, |
| { |
| "baseline_lambda": 0.9999999854707035, |
| "baseline_risk": 0.6439132703267791, |
| "lambda_shift": -0.006853146223264384, |
| "noise": 1.0, |
| "positive_b_lambda": 0.9931468392474391, |
| "positive_b_risk": 0.641600545823265 |
| } |
| ] |
| }, |
| { |
| "assessment": "verified", |
| "baseline_curve": [ |
| 0.32665160920878983, |
| 0.29023776265729345, |
| 0.2787006694510658, |
| 0.27495091856658077, |
| 0.27461167000441256, |
| 0.27603543024839766, |
| 0.2784639008764378, |
| 0.28150052872899933, |
| 0.28491865477647577, |
| 0.2885799876905502, |
| 0.29239587038070275, |
| 0.29630724999374924 |
| ], |
| "baseline_optimal_lambda": 0.05, |
| "baseline_optimal_risk": 0.27461167000441256, |
| "lambda_grid": [ |
| 0.01, |
| 0.02, |
| 0.03, |
| 0.04, |
| 0.05, |
| 0.060000000000000005, |
| 0.06999999999999999, |
| 0.08, |
| 0.09, |
| 0.09999999999999999, |
| 0.11, |
| 0.12 |
| ], |
| "n": 80, |
| "p": 88, |
| "reinforcing_curve": [ |
| 0.3513827776211365, |
| 0.3040007111707276, |
| 0.28523142124036815, |
| 0.2760933518985985, |
| 0.271500405687264, |
| 0.2694512787111104, |
| 0.2689813276125047, |
| 0.2695612414574329, |
| 0.27087304139394297, |
| 0.27271252235267307, |
| 0.2749417002549069, |
| 0.2774635573944301 |
| ], |
| "reinforcing_optimal_lambda": 0.06999999999999999, |
| "reinforcing_optimal_risk": 0.2689813276125047, |
| "released_mechanism": "proportional/perforidge.py, five RRM deployments, Sigma=I, paired seeds", |
| "risk_improvement": 0.00563034239190785, |
| "runs_per_condition": 16 |
| } |
| ], |
| "claims": [ |
| { |
| "claim": 1, |
| "literal_claim": "In the population setting, Theorem 1 characterizes excess risk as a function of the magnitude and direction of the performative effect together with spurious features (Section 4, Theorem 1)." |
| }, |
| { |
| "claim": 2, |
| "literal_claim": "Corollary 2 shows the optimal regularization parameter in the population regime is proportional to the strength of the performative effect, with optimal risk remaining strictly positive (Section 4, Corollary 2)." |
| }, |
| { |
| "claim": 3, |
| "literal_claim": "Theorem 3 establishes a deterministic equivalent of the performative fixed point for over-parameterized ridge regression when the number of features exceeds the number of samples (Section 5, Theorem 3)." |
| }, |
| { |
| "claim": 4, |
| "literal_claim": "Theorem 4 shows the optimal regularization moves in the same direction as the performative effect on predictive features under low noise, but in the opposite direction under high noise, in the over-parameterized regime (Section 5, Theorem 4)." |
| }, |
| { |
| "claim": 5, |
| "literal_claim": "Numerical experiments in Section 6 confirm that in the over-parameterized setting, performative effects can improve optimally-regularized risk when performativity reinforces existing trends, contrasting with the population-regime degradation (Section 6)." |
| } |
| ], |
| "gates": { |
| "all_assessments_decisive": true, |
| "claim1_all_components": true, |
| "claim1_finite": true, |
| "claim2_control_positive": true, |
| "claim2_literal_zero": true, |
| "claim2_proportional": true, |
| "claim3_finite": true, |
| "claim3_gap_within_four_se_plus_finite": true, |
| "claim4_high_noise": true, |
| "claim4_low_noise": true, |
| "claim5_lambda_moves_up": true, |
| "claim5_risk_improves": true |
| }, |
| "paper_id": "G4ve69pimc" |
| } |
|
|