Spaces:
Running
Running
File size: 3,524 Bytes
d70361b | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 | {
"purpose": "Separates pairs used to TUNE calibration (Day 4-6 grid search) from pairs held out for Day 7-8 final evaluation, per the plan Risks sheet mitigation: \"Hold out 20% test pairs never used in grid-search\" (this mitigation was NOT followed during Day 4-6 -- all 16 labeled pairs at the time were used for both tuning AND reported metrics, an in-sample result. This split fixes that going forward.)",
"calibration_set": {
"pair_ids": [
"delhi_0001",
"delhi_0003",
"delhi_0004",
"delhi_0005",
"delhi_0009",
"delhi_0011",
"delhi_0012",
"delhi_0016",
"delhi_0017",
"delhi_0018",
"delhi_0020",
"delhi_0026",
"delhi_0027",
"delhi_0030",
"delhi_0031",
"delhi_0032"
],
"count": 16,
"description": "Used in Day 4 grid search and Day 5/6 verification. The reported +53% F1 lift (0.036 -> 0.055) is measured on THIS set -- in-sample, not a held-out estimate. Do not use for final go/no-go accuracy claims."
},
"held_out_test_set": {
"pair_ids": [
"delhi_0002",
"delhi_0006",
"delhi_0007",
"delhi_0008",
"delhi_0010",
"delhi_0013",
"delhi_0014",
"delhi_0015"
],
"count": 8,
"description": "Labeled Day 7, specifically to never be used in any calibration/grid-search step. All 8 have EMPTY (all-zero) ground truth -- confirmed via visual review (Day 2/3 contact-sheet inspection, re-spot-checked Day 7) as seasonal/crop-texture variation with no real structural change. This makes them a real-imagery false-positive test, analogous to the synthetic brightness_only/parked_cars gates but on genuine Delhi data. CAVEAT: none of these pairs have confirmed real change, so this set tests precision/false-positive behavior only -- it does NOT test recall on held-out real change, since no such pairs were found among the remaining unlabeled batch."
},
"still_unlabeled": {
"pair_ids": [
"delhi_0019",
"delhi_0021",
"delhi_0022",
"delhi_0023",
"delhi_0024",
"delhi_0025",
"delhi_0028",
"delhi_0029"
],
"count": 8,
"description": "Not yet reviewed/labeled. Available for future calibration or held-out use."
},
"held_out_eval_result": {
"date": "Day 7 (2026-07-21)",
"config_tested": "AI-Based Deep Learning, cl_q_base=0.90 (current production default)",
"result": "mean_iou=0.75, mean_f1=0.75 -- 6/8 pairs correctly stayed quiet (IoU=1.0, zero false positives), 2/8 did NOT (mean of 0.75 reflects 6 perfect + 2 pairs scoring exactly 0, not 8 partial scores -- with empty GT, IoU is binary: 1.0 if prediction is also empty, 0.0 if any false positive exists).",
"per_pair": {
"delhi_0002": "ok (quiet)",
"delhi_0006": "ok (quiet)",
"delhi_0007": "FALSE POSITIVE (changePct=0.717%, real Delhi field texture misread as change)",
"delhi_0008": "ok (quiet)",
"delhi_0010": "ok (quiet)",
"delhi_0013": "ok (quiet)",
"delhi_0014": "ok (quiet)",
"delhi_0015": "ok (quiet)"
},
"takeaway": "25% false-positive rate on real held-out Delhi imagery, vs 0% on the clean synthetic brightness_only/parked_cars gates. This is a more realistic (and less flattering) picture than the synthetic-only regression suite gives -- worth reporting alongside the Day 5/6 synthetic PASS results, not instead of them. delhi_0007 is a good specific example to investigate further if false-positive reduction becomes a priority."
}
} |