{ "baseline": { "metrics": { "accuracy": 0.5644171779141104, "average_precision": 0.7310146473284146, "balanced_accuracy": 0.5565888554216868, "brier_score": 0.20382910064244883, "count": 326, "expected_calibration_error": 0.08678660959363686, "f1": 0.22826086956521738, "false_negative_rate": 0.86875, "false_positive_rate": 0.018072289156626505, "fn": 139, "fp": 3, "log_loss": 0.591045865107666, "macro_f1": 0.462421033073207, "negative_count": 166, "positive_count": 160, "precision": 0.875, "recall": 0.13125, "roc_auc": 0.752183734939759, "specificity": 0.9819277108433735, "threshold": 0.7638751511278986, "tn": 163, "tp": 21 }, "name": "acoustic_logistic", "predictions": { "bytes": 71129, "path": "artifacts/partial-baseline/validation_predictions.jsonl", "sha256": "76d67e2c35ac93d9898200f02d8626280b2139a623d432adf1266c9811980626" } }, "bootstrap": { "samples": 2000, "seed": 17, "unit": "audit leakage group" }, "candidate": { "metrics": { "accuracy": 0.5306748466257669, "average_precision": 0.6880689959324726, "balanced_accuracy": 0.5222138554216867, "brier_score": 0.2133037066638398, "count": 326, "expected_calibration_error": 0.1055241439925747, "f1": 0.11560693641618497, "false_negative_rate": 0.9375, "false_positive_rate": 0.018072289156626505, "fn": 150, "fp": 3, "log_loss": 0.6135001064370812, "macro_f1": 0.39809574378220525, "negative_count": 166, "positive_count": 160, "precision": 0.7692307692307693, "recall": 0.0625, "roc_auc": 0.7389683734939759, "specificity": 0.9819277108433735, "threshold": 0.741000771522522, "tn": 163, "tp": 10 }, "name": "tiny_tcn", "predictions": { "bytes": 106299, "path": "artifacts/partial-shard-warmstart-lr3e4-5ep/validation_predictions.jsonl", "sha256": "9eb4c01a9986efc169155c5e132afee9f2b1fe935c05809e10196ccdf6f70ab9" } }, "candidate_minus_baseline_95ci": { "average_precision": { "bootstrap_fraction_delta_above_zero": 0.152, "estimate": -0.042945651395941975, "lower": -0.11945249084849802, "upper": 0.03523632760188885 }, "false_positive_rate": { "bootstrap_fraction_delta_above_zero": 0.425, "estimate": 0.0, "lower": -0.028409090909090908, "upper": 0.02976190476190476 }, "recall": { "bootstrap_fraction_delta_above_zero": 0.011, "estimate": -0.06875, "lower": -0.13297793599323277, "upper": -0.01219512195121951 }, "roc_auc": { "bootstrap_fraction_delta_above_zero": 0.3425, "estimate": -0.013215361445783058, "lower": -0.07823324414295993, "upper": 0.05120708236445977 } }, "count": 326, "fpr_budget": 0.02, "group_count": 326, "negative_count": 166, "positive_count": 160, "scope": "paired development comparison; not independent-test evidence", "threshold_note": "Each threshold was selected on this same development set. Bootstrap samples use those fixed thresholds; intervals do not remove model-selection or calibration bias." }