| { |
| "arxiv_id": "2510.23476", |
| "claims": [ |
| { |
| "assessment": "verified", |
| "claim": 1, |
| "literal_claim": "The optimal collaborative prediction set C*(x) has a two-threshold form: it augments the human-proposed set H(x) with labels whose score exceeds threshold a*, and prunes labels within H(x) whose score exceeds threshold b* (Theorem 2.1, Section 2).", |
| "result": { |
| "exhaustive_instances": 64, |
| "inclusion_rule_disambiguation": "The paper uses nonconformity s=1-p and includes s<=threshold; equivalently it includes posterior p>=1-threshold. Outside-H passing a augments, while inside-H failing b is pruned.", |
| "one_threshold_suboptimal_instances": 20, |
| "policies_enumerated": 32768, |
| "two_threshold_zero_gap_instances": 64 |
| } |
| }, |
| { |
| "assessment": "verified", |
| "claim": 2, |
| "literal_claim": "The counterfactual harm principle requires P(Y not in C(X) | Y in H(X)) < epsilon, i.e., the AI must not remove correct labels the human already proposed (Section 1).", |
| "result": { |
| "epsilon": "1/5", |
| "event_computed": "Y not in C(X), conditioned on Y in H(X)", |
| "exact_counterfactual_harm": "2950/39211", |
| "prune_all_control_harm": "1/1" |
| } |
| }, |
| { |
| "assessment": "verified", |
| "claim": 3, |
| "literal_claim": "The complementarity principle requires P(Y in C(X) | Y not in H(X)) >= 1-delta, i.e., the AI must recover correct labels the human missed (Section 1).", |
| "result": { |
| "delta": "1/4", |
| "event_computed": "Y in C(X), conditioned on Y not in H(X)", |
| "exact_complementarity": "17851/23308", |
| "no_augmentation_control_complementarity": "0/1" |
| } |
| }, |
| { |
| "assessment": "verified", |
| "claim": 4, |
| "literal_claim": "The offline calibration procedure gives distribution-free finite-sample coverage guarantees whose bounds depend on the sizes of the calibration sets used to fit a* and b* (Proposition 4.1, Section 4.1).", |
| "result": { |
| "all_lower_bounds_hold": true, |
| "all_strict_upper_bounds_hold": true, |
| "exact_rank_cases": 766, |
| "floor_quantile_control_violations": 3, |
| "group_size_settings": 14 |
| } |
| }, |
| { |
| "assessment": "verified", |
| "claim": 5, |
| "literal_claim": "The CUP-Online algorithm updates thresholds a_t and b_t via stochastic approximation steps (a_{t+1}=a_t+eta(1{s(x_t,y_t)>a_t}-delta), and analogously for b_t) to track target error rates under arbitrary distribution shift, including shifts caused by humans adapting their behavior to the AI (Section 4.2, Proposition 4.2).", |
| "result": { |
| "all_claimed_bounds_hold": true, |
| "all_exact_telescopes": true, |
| "correct_update_rounds": 76000, |
| "correct_update_trajectories": 16, |
| "delta": "1/5", |
| "empirical_log_deviation_slope": -1.0459650194247514, |
| "epsilon": "1/10", |
| "eta": "1/5", |
| "rate_horizons": [ |
| 1000, |
| 2000, |
| 4000, |
| 12000 |
| ], |
| "rate_repetitions_per_horizon": 8, |
| "sign_flipped_control_bound_violations": 2 |
| } |
| } |
| ], |
| "execution": { |
| "deterministic": true, |
| "exact_rational_arithmetic": true, |
| "nearby_proxy_used": false, |
| "peer_artifacts_used": false, |
| "source_only_verdicts": false |
| }, |
| "paper_id": "FzP6XZGG4d", |
| "schema": "icml-native-reproduction-v1" |
| } |
|
|