d1a-e4b / result.json
JohnP1's picture
D1A-E4B v0.5: Kev's skills back on top of v0.4 (hard 55->71, devtools 64->70); PR blast/severity trade-off in the card
e7db32f verified
Raw History Blame Contribute Delete
3.93 kB
{
"model": "D1A-E4B v0.5",
"temperature": 1.782,
"method": "d1a.eval.benchmark through scripts/quality_gate.py (--head-run, --card-suites, --suites) on the MLX 8-bit builds of v0.4 and v0.5, held-out partitions; pr-labels:test and test-ja there cover only the records inside the training context (183 and 18). The full PR test sets were scored on the bf16 PyTorch checkpoints with the serving context.",
"sets": {
"devtools-v1": {
"v0.4": {
"n": 1074,
"acc": 0.6378,
"ece": 0.1119,
"nll": 0.8652,
"mean_conf": 0.7451
},
"v0.5": {
"n": 1074,
"acc": 0.6983,
"ece": 0.0483,
"nll": 0.6713,
"mean_conf": 0.6951
}
},
"documents-v1": {
"v0.4": {
"n": 920,
"acc": 0.8663,
"ece": 0.0297,
"nll": 0.3864,
"mean_conf": 0.8838
},
"v0.5": {
"n": 920,
"acc": 0.875,
"ece": 0.0417,
"nll": 0.3556,
"mean_conf": 0.9135
}
},
"hard-v1": {
"v0.4": {
"n": 1083,
"acc": 0.5522,
"ece": 0.0379,
"nll": 1.0292,
"mean_conf": 0.559
},
"v0.5": {
"n": 1083,
"acc": 0.7147,
"ece": 0.0448,
"nll": 0.6907,
"mean_conf": 0.6707
}
},
"ja-jglue_development": {
"v0.4": {
"n": 1500,
"acc": 0.81,
"ece": 0.0299,
"nll": 0.4757,
"mean_conf": 0.7831
},
"v0.5": {
"n": 1500,
"acc": 0.8207,
"ece": 0.0263,
"nll": 0.4357,
"mean_conf": 0.7996
}
},
"pr-labels_test": {
"v0.4": {
"n": 393,
"acc": 0.8219,
"ece": 0.0292,
"nll": 0.4943,
"mean_conf": 0.8179
},
"v0.5": {
"n": 393,
"acc": 0.8499,
"ece": 0.0436,
"nll": 0.4933,
"mean_conf": 0.8207
}
},
"pr-labels_test-ja": {
"v0.4": {
"n": 37,
"acc": 0.8649,
"ece": 0.0772,
"nll": 0.5611,
"mean_conf": 0.8218
},
"v0.5": {
"n": 37,
"acc": 0.8649,
"ece": 0.0775,
"nll": 0.5395,
"mean_conf": 0.8212
}
},
"routing_factory-development": {
"v0.4": {
"n": 640,
"acc": 0.9078,
"ece": 0.073,
"nll": 0.2972,
"mean_conf": 0.835
},
"v0.5": {
"n": 640,
"acc": 0.9047,
"ece": 0.1301,
"nll": 0.3357,
"mean_conf": 0.7806
}
},
"routing_generic-development": {
"v0.4": {
"n": 270,
"acc": 0.9741,
"ece": 0.0219,
"nll": 0.074,
"mean_conf": 0.9631
},
"v0.5": {
"n": 270,
"acc": 0.9889,
"ece": 0.0287,
"nll": 0.0555,
"mean_conf": 0.9602
}
},
"routing_handlabelled-45": {
"v0.4": {
"n": 45,
"acc": 1.0,
"ece": 0.0282,
"nll": 0.0307,
"mean_conf": 0.9718
},
"v0.5": {
"n": 45,
"acc": 1.0,
"ece": 0.0323,
"nll": 0.0347,
"mean_conf": 0.9677
}
},
"v4_transfer-v4": {
"v0.4": {
"n": 656,
"acc": 0.689,
"ece": 0.1276,
"nll": 0.735,
"mean_conf": 0.8166
},
"v0.5": {
"n": 656,
"acc": 0.7165,
"ece": 0.1059,
"nll": 0.7034,
"mean_conf": 0.809
}
},
"v7_decision-v7": {
"v0.4": {
"n": 1264,
"acc": 0.8465,
"ece": 0.0262,
"nll": 0.4252,
"mean_conf": 0.8695
},
"v0.5": {
"n": 1264,
"acc": 0.8528,
"ece": 0.0299,
"nll": 0.4026,
"mean_conf": 0.882
}
}
},
"pr_labels_full_bf16": {
"test": {
"v0.4": {
"type": 0.877,
"sev": 0.778,
"blast": 0.54,
"all": 0.8078
},
"v0.5": {
"type": 0.897,
"sev": 0.756,
"blast": 0.583,
"all": 0.8098
},
"n_prs": 953
},
"test-ja": {
"v0.4": {
"all": 0.7409
},
"v0.5": {
"all": 0.7461
},
"n_prs": 92
}
},
"owner_prs_hand_checked": {
"n_prs": 39,
"v0.4": {
"type": 0.868,
"blast": 0.872,
"sev": 0.656
},
"v0.5": {
"type": 0.842,
"blast": 0.744,
"sev": 0.656
}
},
"labeler_replay_review_bot": {
"n": 98,
"v0.4": {
"type": 0.863,
"blast": 0.704
},
"v0.5": {
"type": 0.863,
"blast": 0.633
}
}
}