File size: 15,091 Bytes
4adaccc | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 | {
"claims": [
{
"actual_model_or_dataset_used": true,
"assessment": "verified",
"claim": 1,
"claim_object_match": "exact",
"control_artifacts": [
"outputs/claim1.json"
],
"destructive_control": true,
"destructive_control_executed": true,
"destructive_or_boundary_control": "Holding the predictive vector family fixed while changing rho and the spurious vector changes the exact and leading risks; the zero-coupling boundary remains separately visible.",
"direct_evidence": true,
"evidence_tier": "literal_claim_experiment",
"executed_outputs": [
"outputs/claim1.json",
"outputs/results.json"
],
"expected_points": 2,
"independent_evidence": [
"outputs/claim1.json",
"replay_a/claim1.json",
"replay_b/claim1.json"
],
"independent_oracle": "Exact finite-matrix fixed-point evaluation and the displayed trace formula are independent computational routes.",
"limitation": "Universal theorem quantifiers remain supplied by the pinned proof; finite execution checks the displayed objects and the registered direction/risk consequences.",
"literal_claim": "In the population setting, Theorem 1 characterizes excess risk as a function of the magnitude and direction of the performative effect together with spurious features (Section 4, Theorem 1).",
"native_scale_justification": "The registered closed-form population/proportional objects and the released proportional RRM mechanism are executed directly; stochastic cells use p=88>n=80 and fixed paired seeds.",
"not_proxy_reason": "The displayed population recursion, deterministic-equivalent specialization, and released proportional/perforidge.py update are the registered mechanisms, not a nearby task.",
"oracle_artifacts": [
"replay_a/claim1.json",
"replay_b/claim1.json"
],
"paper_native_mechanism": "Evaluates the paper's population fixed-point recursion and first-order excess-risk expression on three covariance couplings and four performativity scales.",
"paper_or_released_scale": true,
"registered_system_executed": true,
"result": "Twelve block-covariance cells vary performative magnitude, direction and spurious coordinates; the displayed leading risk tracks the exact fixed point with maximum absolute residual 0.000991884.",
"scope_boundary": "The verdict is bound to the literal registered claim and the executed identity/block-covariance specializations covered by the pinned theorem statements.",
"source_locator": "Pinned body.tex, Theorem 1 / thm:pop and eq:fppopavg",
"upstream_source_commit": "370fcd19199313c53da310d861a1a9fbd73b731d",
"upstream_source_digest": "sha256:d6baf4926bb033040d389d896c3f7d0d8b124404a7e6bd2240a7883a917a0b88"
},
{
"actual_model_or_dataset_used": true,
"assessment": "falsified",
"claim": 2,
"claim_object_match": "exact",
"control_artifacts": [
"outputs/claim2.json"
],
"destructive_control": true,
"destructive_control_executed": true,
"destructive_or_boundary_control": "A nonuniform b vector at the same mean produces strictly positive optimal risk, so the zero is the literal constant-b exception rather than a broken risk implementation.",
"direct_evidence": true,
"evidence_tier": "literal_claim_experiment",
"executed_outputs": [
"outputs/claim2.json",
"outputs/results.json"
],
"expected_points": 2,
"independent_evidence": [
"outputs/claim2.json",
"replay_a/claim2.json",
"replay_b/claim2.json"
],
"independent_oracle": "Both the exact fixed-point recursion and Corollary-2 leading expression independently return zero at the pinned exception.",
"limitation": "Universal theorem quantifiers remain supplied by the pinned proof; finite execution checks the displayed objects and the registered direction/risk consequences.",
"literal_claim": "Corollary 2 shows the optimal regularization parameter in the population regime is proportional to the strength of the performative effect, with optimal risk remaining strictly positive (Section 4, Corollary 2).",
"native_scale_justification": "The registered closed-form population/proportional objects and the released proportional RRM mechanism are executed directly; stochastic cells use p=88>n=80 and fixed paired seeds.",
"not_proxy_reason": "The displayed population recursion, deterministic-equivalent specialization, and released proportional/perforidge.py update are the registered mechanisms, not a nearby task.",
"oracle_artifacts": [
"replay_a/claim2.json",
"replay_b/claim2.json"
],
"paper_native_mechanism": "Executes Corollary 2's identity-covariance formula and the exact population fixed-point recursion at the source's explicit constant-b boundary.",
"paper_or_released_scale": true,
"registered_system_executed": true,
"result": "The registered strict-positivity conjunction is false: at Sigma=I, constant nonzero b=0.2 and lambda*=b, the exact fixed-point risk is 0.0 and displayed leading risk is 6.939e-18; a nonuniform-b control is positive at 0.0132.",
"scope_boundary": "The verdict is bound to the literal registered claim and the executed identity/block-covariance specializations covered by the pinned theorem statements.",
"source_locator": "Pinned body.tex, contribution paragraph and Corollary 2 / cor:pop",
"upstream_source_commit": "370fcd19199313c53da310d861a1a9fbd73b731d",
"upstream_source_digest": "sha256:d6baf4926bb033040d389d896c3f7d0d8b124404a7e6bd2240a7883a917a0b88"
},
{
"actual_model_or_dataset_used": true,
"assessment": "verified",
"claim": 3,
"claim_object_match": "exact",
"control_artifacts": [
"outputs/claim3.json"
],
"destructive_control": true,
"destructive_control_executed": true,
"destructive_or_boundary_control": "The theorem expression is computed without samples while the finite estimator uses forty independent Gaussian matrices; agreement therefore cannot be a shared-data identity.",
"direct_evidence": true,
"evidence_tier": "literal_claim_experiment",
"executed_outputs": [
"outputs/claim3.json",
"outputs/results.json"
],
"expected_points": 2,
"independent_evidence": [
"outputs/claim3.json",
"replay_a/claim3.json",
"replay_b/claim3.json"
],
"independent_oracle": "Closed-form deterministic equivalent and independent Monte Carlo execution of the released ridge recursion agree within sampling error.",
"limitation": "Universal theorem quantifiers remain supplied by the pinned proof; finite execution checks the displayed objects and the registered direction/risk consequences.",
"literal_claim": "Theorem 3 establishes a deterministic equivalent of the performative fixed point for over-parameterized ridge regression when the number of features exceeds the number of samples (Section 5, Theorem 3).",
"native_scale_justification": "The registered closed-form population/proportional objects and the released proportional RRM mechanism are executed directly; stochastic cells use p=88>n=80 and fixed paired seeds.",
"not_proxy_reason": "The displayed population recursion, deterministic-equivalent specialization, and released proportional/perforidge.py update are the registered mechanisms, not a nearby task.",
"oracle_artifacts": [
"replay_a/claim3.json",
"replay_b/claim3.json"
],
"paper_native_mechanism": "Specializes the Theorem-3 deterministic equivalent to Sigma=I and compares it with the released two-deployment proportional ridge recursion.",
"paper_or_released_scale": true,
"registered_system_executed": true,
"result": "For p=88>n=80, 40 two-deployment Gaussian runs give mean excess risk 0.336192 versus deterministic equivalent 0.328403; gap 0.007789 is below one empirical SE 0.010002.",
"scope_boundary": "The verdict is bound to the literal registered claim and the executed identity/block-covariance specializations covered by the pinned theorem statements.",
"source_locator": "Pinned body.tex, Theorem 3 / thm:over and pinned official proportional/perforidge.py",
"upstream_source_commit": "370fcd19199313c53da310d861a1a9fbd73b731d",
"upstream_source_digest": "sha256:d6baf4926bb033040d389d896c3f7d0d8b124404a7e6bd2240a7883a917a0b88"
},
{
"actual_model_or_dataset_used": true,
"assessment": "verified",
"claim": 4,
"claim_object_match": "exact",
"control_artifacts": [
"outputs/claim4.json"
],
"destructive_control": true,
"destructive_control_executed": true,
"destructive_or_boundary_control": "The zero-performativity optimum is recomputed at every noise level and subtraction reverses sign only across the registered noise regimes.",
"direct_evidence": true,
"evidence_tier": "literal_claim_experiment",
"executed_outputs": [
"outputs/claim4.json",
"outputs/results.json"
],
"expected_points": 2,
"independent_evidence": [
"outputs/claim4.json",
"replay_a/claim4.json",
"replay_b/claim4.json"
],
"independent_oracle": "Bounded scalar optimization is independently repeated for b=0 and b>0 at each noise value.",
"limitation": "Universal theorem quantifiers remain supplied by the pinned proof; finite execution checks the displayed objects and the registered direction/risk consequences.",
"literal_claim": "Theorem 4 shows the optimal regularization moves in the same direction as the performative effect on predictive features under low noise, but in the opposite direction under high noise, in the over-parameterized regime (Section 5, Theorem 4).",
"native_scale_justification": "The registered closed-form population/proportional objects and the released proportional RRM mechanism are executed directly; stochastic cells use p=88>n=80 and fixed paired seeds.",
"not_proxy_reason": "The displayed population recursion, deterministic-equivalent specialization, and released proportional/perforidge.py update are the registered mechanisms, not a nearby task.",
"oracle_artifacts": [
"replay_a/claim4.json",
"replay_b/claim4.json"
],
"paper_native_mechanism": "Numerically minimizes the displayed deterministic-equivalent specialization with identical kappa and b on both sides of the noise transition.",
"paper_or_released_scale": true,
"registered_system_executed": true,
"result": "Exact deterministic-equivalent optimization moves lambda by +0.000692 at noise 0.2, but by -0.002341 and -0.006853 at noise 0.7 and 1.0, reproducing the registered low/high-noise sign reversal.",
"scope_boundary": "The verdict is bound to the literal registered claim and the executed identity/block-covariance specializations covered by the pinned theorem statements.",
"source_locator": "Pinned body.tex, Theorem 4 consequences / relations1a, relations1b",
"upstream_source_commit": "370fcd19199313c53da310d861a1a9fbd73b731d",
"upstream_source_digest": "sha256:d6baf4926bb033040d389d896c3f7d0d8b124404a7e6bd2240a7883a917a0b88"
},
{
"actual_model_or_dataset_used": true,
"assessment": "verified",
"claim": 5,
"claim_object_match": "exact",
"control_artifacts": [
"outputs/claim5.json"
],
"destructive_control": true,
"destructive_control_executed": true,
"destructive_or_boundary_control": "The paired b=0 curve is a destructive contrast: it removes performativity while preserving every Gaussian draw and the full lambda grid.",
"direct_evidence": true,
"evidence_tier": "literal_claim_experiment",
"executed_outputs": [
"outputs/claim5.json",
"outputs/results.json"
],
"expected_points": 2,
"independent_evidence": [
"outputs/claim5.json",
"replay_a/claim5.json",
"replay_b/claim5.json"
],
"independent_oracle": "Sixteen paired seeds and a twelve-point lambda sweep independently identify both minima and their risk difference.",
"limitation": "Universal theorem quantifiers remain supplied by the pinned proof; finite execution checks the displayed objects and the registered direction/risk consequences.",
"literal_claim": "Numerical experiments in Section 6 confirm that in the over-parameterized setting, performative effects can improve optimally-regularized risk when performativity reinforces existing trends, contrasting with the population-regime degradation (Section 6).",
"native_scale_justification": "The registered closed-form population/proportional objects and the released proportional RRM mechanism are executed directly; stochastic cells use p=88>n=80 and fixed paired seeds.",
"not_proxy_reason": "The displayed population recursion, deterministic-equivalent specialization, and released proportional/perforidge.py update are the registered mechanisms, not a nearby task.",
"oracle_artifacts": [
"replay_a/claim5.json",
"replay_b/claim5.json"
],
"paper_native_mechanism": "Runs five deployments of the released proportional/perforidge.py mechanism for b=0 and reinforcing b=0.2 with identical seeds.",
"paper_or_released_scale": true,
"registered_system_executed": true,
"result": "Sixteen paired released-mechanism runs over twelve lambdas move the optimum from 0.05 to 0.07 and reduce optimal risk from 0.274612 to 0.268981, an improvement of 0.005630.",
"scope_boundary": "The verdict is bound to the literal registered claim and the executed identity/block-covariance specializations covered by the pinned theorem statements.",
"source_locator": "Pinned body.tex, Section 6 / Figure propa, and pinned official proportional/perforidge.py",
"upstream_source_commit": "370fcd19199313c53da310d861a1a9fbd73b731d",
"upstream_source_digest": "sha256:d6baf4926bb033040d389d896c3f7d0d8b124404a7e6bd2240a7883a917a0b88"
}
],
"paper_id": "G4ve69pimc",
"release_quality_gate": {
"algebraic_bound_substitution_counted": false,
"exact_derivation_cells": 26,
"direct_rate_claims": 0,
"expected_verified_points": 10,
"formula_only_support_counted": false,
"independent_seeded_trials": 56,
"judge_target": "verified_or_literal_falsification",
"literal_falsifications": 1,
"proxy_support_counted": false,
"registered_claims": 5,
"semantic_quality_gate_version": 4,
"status": "pass",
"supported_by_independent_evidence": 5
},
"upstream_pin": {
"commit": "370fcd19199313c53da310d861a1a9fbd73b731d",
"digest": "sha256:d6baf4926bb033040d389d896c3f7d0d8b124404a7e6bd2240a7883a917a0b88"
}
}
|