qev / release-comparison /input-protocol.json
ken-jo's picture
Release QEV 0.1.1: Qwen3.5-2B typed decisions
f996835 verified
Raw History Blame Contribute Delete
4.34 kB
{
"version": "qev-0.1.1-final-comparison",
"official_dataset_revision": "d51d993547ad8355b1c25157fbc1fea0649e8ffa",
"official_dataset_sha256": "4881baae1cfd752311a58064cb2095c0457ee2c0b8914d1cfe2590e19933e7b3",
"sources": {
"ag_news": {
"repo": "fancyzhx/ag_news",
"revision": "eb185aade064a813bc0b7f42de02595523103ca4",
"file": "data/test-00000-of-00001.parquet",
"criteria": {
"world": "world news and international politics",
"sports": "sports",
"business": "business and economy",
"sci_tech": "science and technology"
},
"field": "article",
"question": "topic",
"instructions": "What is the topic of `article`?",
"laya_training_status": "In training according to upstream bench_apps.py",
"download_sha256": "71de87ec66bc5737752a2502204dfa6d7fe9856ade3ea444dc6317789a4f13fb",
"selected": "first 400 test rows",
"class_counts": {
"2": 72,
"3": 102,
"1": 123,
"0": 103
}
},
"emotion": {
"repo": "dair-ai/emotion",
"revision": "cab853a1dbdf4c42c2b3ef2173804746df8825fe",
"file": "split/test-00000-of-00001.parquet",
"criteria": {
"sadness": "sadness",
"joy": "joy",
"love": "love",
"anger": "anger",
"fear": "fear",
"surprise": "surprise"
},
"field": "text",
"question": "emotion",
"instructions": "Which emotion is most strongly expressed in `text`?",
"laya_training_status": "Held out according to upstream bench_apps.py",
"download_sha256": "6f8407fa1ca9c310f55781f082ed73812f6551e8dda2c61973123a121869245b",
"selected": "first 400 test rows",
"class_counts": {
"0": 119,
"1": 118,
"4": 52,
"3": 64,
"2": 37,
"5": 10
}
}
},
"requests_sha256": "585850d575d9dbdf943683936dda7e07b9e5525b60875906926e9915c557d1d6",
"questions": {
"typed_decisions": 2000,
"ag_news": 400,
"emotion": 400
},
"script_sha256": "83bcc709680109b9c9f4c86a84e7c7659f34a30408de90af2695a6d1924edd0b",
"adaptation_data_audit": {
"corpora": 15,
"rows_including_repeated_stages": 251457,
"source_counts": {
"veyra-original-generator": 5472,
"AI-Lab-Makerere/beans": 8312,
"veyra-policy-generator-v3": 8384,
"veyra-evidence-interventions-v6": 2400,
"veyra-policy-refresh-v8": 20294,
"PolyAI/banking77": 24132,
"veyra/conditional-probability": 480,
"stanfordnlp/snli": 23220,
"vminhkhoi/trashnet": 16683,
"veyra-workflow-v12": 70560,
"LocalLLaMA/typed-decisions": 49200,
"veyra-workspace-calibration-v12": 4080,
"workflow-depth-policy-fit-data-v12": 4080,
"workflow-depth-policy-validation-data-v12": 4080,
"workflow-cohort-policy-validation-1-v12": 2040,
"workflow-cohort-policy-validation-2-v12": 2040,
"workflow-recovery-policy-validation-1-v13": 2040,
"workflow-recovery-policy-validation-2-v13": 2040,
"workflow-recovery-fresh-final-v13": 1920
},
"normalized_substring_matches": [],
"scope": "Released adaptation snapshots; unknown backbone pretraining overlap"
},
"selection": "All 2000 typed decisions; first 400 AG News and Emotion test rows",
"upstream_alignment": "Application sample count, row selection and instructions follow LAYA bench_apps.py. Emotion null descriptions become label names for both SDKs. State JSON is identical.",
"evaluation": {
"models": [
"qev",
"laya-base",
"laya-specialist"
],
"one_question_per_call": true,
"laya_max_len": 1024,
"qev_max_tokens": 2048,
"warmups_per_suite": 3,
"torch_threads": 4,
"device": "Single RTX 4060 Ti 8 GB, models run sequentially",
"no_parameter_or_temperature_or_prompt_selection": true,
"headline_accuracy": "All questions including abstained answers, explicit hard gold",
"probability_metrics": "Same frozen decision_metrics.py for all models",
"zero_shot": "No task-specific QEV adaptation or examples; not pretraining-clean",
"typed_decisions": "Both QEV and specialist adapted; reused regression benchmark"
}
}