| { |
| "schema_version": "1.0", |
| "title": "Reproduction: Conditional Coverage Diagnostics for Conformal Prediction", |
| "space_id": "snaykey/repro-conditional-coverage-diagnostics", |
| "paper": { |
| "arxiv_id": "2512.11779", |
| "openreview_id": "vaApZm6MKM" |
| }, |
| "tags": [ |
| "icml2026-repro", |
| "paper-vaApZm6MKM" |
| ], |
| "updated_at": "2026-07-29T11:00:00+00:00", |
| "root": { |
| "slug": "index", |
| "title": "Reproduction: Conditional Coverage Diagnostics for Conformal Prediction", |
| "children": [ |
| { |
| "slug": "executive-summary", |
| "title": "Executive summary", |
| "children": [] |
| }, |
| { |
| "slug": "claim-1-ert-family", |
| "title": "The paper introduces a family of Excess Risk of the Target Coverage (ERT) metrics—L1-ERT, L2-ERT, and KL-ERT—for evaluating conditional coverage in conformal prediction, built on the principle that no classifier can outperform a constant 1-alpha predictor under perfect conditional coverage (Table 1, Section 3.1).", |
| "children": [] |
| }, |
| { |
| "slug": "claim-2-classifier-power", |
| "title": "Using LightGBM as the underlying classifier for L1-ERT achieves 68.4% relative statistical power (relative to maximum), compared to only 38.3% for the PartitionWise classifier underlying CovGap (Table 2, Section 4.1).", |
| "children": [] |
| }, |
| { |
| "slug": "claim-3-sample-efficiency", |
| "title": "In synthetic experiments, group-based metrics like CovGap remain unreliable and unaligned with their theoretical values even at 5,000 test points, whereas ERT metrics such as L1-ERT converge with far fewer samples (Figure 4, Section 4.2).", |
| "children": [] |
| }, |
| { |
| "slug": "claim-4-asymmetric-components", |
| "title": "The metrics decompose conditional coverage error into asymmetric components ell_plus-ERT and ell_minus-ERT, separating unnecessary conservatism (over-coverage) from excessive aggressiveness (under-coverage) (Section 3.3).", |
| "children": [] |
| }, |
| { |
| "slug": "claim-5-classification-decomposition", |
| "title": "Classification experiments report divergent KL_plus-ERT and KL_minus-ERT values across conformal prediction methods, demonstrating the over/under-coverage decomposition in practice (Table 4, Section 4.3.2).", |
| "children": [] |
| }, |
| { |
| "slug": "claim-6-cross-validation", |
| "title": "Algorithm 1 estimates the ERT metrics from finite samples using k-fold cross-validation to avoid overfitting the classifier used in the estimation (Algorithm 1).", |
| "children": [] |
| }, |
| { |
| "slug": "conclusion", |
| "title": "Conclusion", |
| "children": [] |
| } |
| ] |
| }, |
| "revision": 1 |
| } |
|
|