Reza2kn's picture
Lower Mac streaming batch size after separate paired qualification
52ecfff verified
Raw History Blame Contribute Delete
7.42 kB
{
"status": "selected_for_fresh_qualification",
"selected_batch": 2,
"selected_source_root": "work/mlx-native-q3-v1",
"weights_unchanged": true,
"selection_data": "Exposed AMI IS1000a only; reserved11 results not reused for selection",
"cases": [
{
"batch": 4,
"report": "/Users/Ajab/conductor/workspaces/visualears/experiments/audio8-ternary/reports/mlx-native-q3-stream-v1/verification.json",
"report_sha256": "0d9e39bf29670fd54631b693e51a0cd8873105989a47239b588d0273a53255cc",
"generated_ids_sha256": "72f994120072e2b864ea8de08e44195efe481d10a74dedf2e78bd4a1a5eea4b3",
"word_errors": {
"hits": 303,
"substitutions": 39,
"deletions": 136,
"insertions": 4,
"reference_units": 478,
"hypothesis_units": 346,
"errors": 179,
"error_rate": 0.37447698744769875
},
"matched_word_timing": {
"measured": true,
"actual_arrival_paced": true,
"timing_basis": "Independent host receipt of flushed JSONL since paced20ms producer start; all source words retained",
"correctly_matched_units": 303,
"ambiguous_matched_units": 9,
"ambiguity_method": "all minimum unit-cost Levenshtein paths, without timing-based selection",
"unambiguous_matched_units": 294,
"unambiguous_maximum_seconds": 5.586898207997422,
"unmatched_reference_units": 175,
"matched_reference_fraction": 0.6338912133891214,
"min_seconds": -2.193736457971383,
"p50_seconds": 0.6946411670209045,
"p95_seconds": 0.8736333750304794,
"maximum_seconds": 5.586898207997422,
"matched_units_above_one_second": 4,
"all_matched_units_within_one_second": null,
"whole_release_pass": false,
"limitation": "Lag summaries use one deterministic text alignment and are descriptive when alternative optimal alignments exist. Only unambiguous matches can support per-word timing claims. Omissions/substitutions remain errors, never zero-latency successes. Published reference timing uncertainty and overlapping-speaker word order are not independently corrected."
},
"causal_timing_audit": {
"count": 293,
"p95_seconds": 0.8712869580299412,
"max_seconds": 1.1078053750237586,
"above_one_second": 2
},
"cross_reset_matches": 1,
"mature_work_rtf": 0.7764502045640678
},
{
"batch": 2,
"report": "/Users/Ajab/conductor/workspaces/visualears/experiments/audio8-ternary/reports/mlx-native-q3-b2-exposed-v1/run.json",
"report_sha256": "789a6658c508b70aba1060c3cdd0412b87bd7b35e4793d47f8cb45e4c46c61e5",
"generated_ids_sha256": "3d3130b78aad7aa764dfbf1aeae3591847f84f89379f03fb6eab9a072cb6d18c",
"word_errors": {
"hits": 302,
"substitutions": 39,
"deletions": 137,
"insertions": 3,
"reference_units": 478,
"hypothesis_units": 344,
"errors": 179,
"error_rate": 0.37447698744769875
},
"matched_word_timing": {
"measured": true,
"actual_arrival_paced": true,
"timing_basis": "Independent host receipt of flushed JSONL since paced20ms producer start; all source words retained",
"correctly_matched_units": 302,
"ambiguous_matched_units": 8,
"ambiguity_method": "all minimum unit-cost Levenshtein paths, without timing-based selection",
"unambiguous_matched_units": 294,
"unambiguous_maximum_seconds": 5.489463750023845,
"unmatched_reference_units": 176,
"matched_reference_fraction": 0.6317991631799164,
"min_seconds": -2.2680539579666146,
"p50_seconds": 0.5616992500261517,
"p95_seconds": 0.7167164999945186,
"maximum_seconds": 5.489463750023845,
"matched_units_above_one_second": 1,
"all_matched_units_within_one_second": null,
"whole_release_pass": false,
"limitation": "Lag summaries use one deterministic text alignment and are descriptive when alternative optimal alignments exist. Only unambiguous matches can support per-word timing claims. Omissions/substitutions remain errors, never zero-latency successes. Published reference timing uncertainty and overlapping-speaker word order are not independently corrected."
},
"causal_timing_audit": {
"count": 293,
"p95_seconds": 0.7042127919802397,
"max_seconds": 0.9364958749921044,
"above_one_second": 0
},
"cross_reset_matches": 1,
"mature_work_rtf": 0.8644507736555196
},
{
"batch": 3,
"report": "/Users/Ajab/conductor/workspaces/visualears/experiments/audio8-ternary/reports/mlx-native-q3-b3-exposed-v2/run.json",
"report_sha256": "ac20ccdcf5eb13b28227de3428465c6caac31f1e21032b2037b16308f43e5cd3",
"generated_ids_sha256": "3d3130b78aad7aa764dfbf1aeae3591847f84f89379f03fb6eab9a072cb6d18c",
"word_errors": {
"hits": 302,
"substitutions": 39,
"deletions": 137,
"insertions": 3,
"reference_units": 478,
"hypothesis_units": 344,
"errors": 179,
"error_rate": 0.37447698744769875
},
"matched_word_timing": {
"measured": true,
"actual_arrival_paced": true,
"timing_basis": "Independent host receipt of flushed JSONL since paced20ms producer start; all source words retained",
"correctly_matched_units": 302,
"ambiguous_matched_units": 8,
"ambiguity_method": "all minimum unit-cost Levenshtein paths, without timing-based selection",
"unambiguous_matched_units": 294,
"unambiguous_maximum_seconds": 5.515487207956596,
"unmatched_reference_units": 176,
"matched_reference_fraction": 0.6317991631799164,
"min_seconds": -2.1270811670180336,
"p50_seconds": 0.6239733329950923,
"p95_seconds": 0.7859231659909938,
"maximum_seconds": 5.515487207956596,
"matched_units_above_one_second": 4,
"all_matched_units_within_one_second": null,
"whole_release_pass": false,
"limitation": "Lag summaries use one deterministic text alignment and are descriptive when alternative optimal alignments exist. Only unambiguous matches can support per-word timing claims. Omissions/substitutions remain errors, never zero-latency successes. Published reference timing uncertainty and overlapping-speaker word order are not independently corrected."
},
"causal_timing_audit": {
"count": 293,
"p95_seconds": 0.7846965410071789,
"max_seconds": 1.0496153329871731,
"above_one_second": 2
},
"cross_reset_matches": 1,
"mature_work_rtf": 0.8001077809603885
}
],
"reason": "Batch2 and batch3 have identical tokens. All three batches have179/478 errors, but batch2/3 have one more deletion and one fewer insertion than batch4. Batch2 has the lowest delay and no >1 second same-session unambiguous match in this exposed diagnostic while mature RTF remains below1. Cross-reset textual matches remain unresolved recognition evidence, not latency successes. No universal pass inferred.",
"new_inputs": [
"data/continuous-v3/manifest.json",
"data/heldout-zh-v3/manifest.json"
],
"selector_sha256": "7ba7cbed4125fd7834a57ba909ca464b4d970e39487b551b567e0b4b9265acd0",
"production_ready": false
}