coneml-810m-alpha / eval /summary.json
RandomMountainMan's picture
Publish signed-off ConeML 810M release
61e2c88 verified
Raw
History Blame Contribute Delete
4.3 kB
{
"schema_version": 1,
"model": "ConeML/coneml-810m-alpha",
"scope": "Aggregate results from ConeML internal held-out instruments and the stated public-benchmark subsets. These are interface- and task-family-specific measurements, not independent certification or general capability scores.",
"bf16": {
"task_family_certification": {
"ckpt": "ConeML/coneml-810m-alpha",
"decoding": "greedy, rep_penalty 1.15, frame 'Question:/Answer:'",
"by_category": {
"add-1d": {
"correct": 70,
"n": 71,
"acc": 0.9859
},
"add-2d-carry": {
"correct": 266,
"n": 500,
"acc": 0.532
},
"comparison": {
"correct": 363,
"n": 500,
"acc": 0.726
},
"missing-addend": {
"correct": 358,
"n": 500,
"acc": 0.716
},
"missing-factor": {
"correct": 89,
"n": 500,
"acc": 0.178
},
"mul-1d": {
"correct": 44,
"n": 45,
"acc": 0.9778
},
"mul-2dx1d": {
"correct": 487,
"n": 500,
"acc": 0.974
},
"sub-borrow": {
"correct": 455,
"n": 500,
"acc": 0.91
},
"two-step": {
"correct": 395,
"n": 500,
"acc": 0.79
},
"which-bigger": {
"correct": 285,
"n": 300,
"acc": 0.95
}
}
},
"executed_single_function_writing": {
"ckpt": "ConeML/coneml-810m-alpha",
"passed": 207,
"n": 300,
"rate": 0.69,
"by_family": {
"code_arith": "18/24",
"code_branch": "11/31",
"code_count": "34/35",
"code_dedup": "20/26",
"code_filter": "2/25",
"code_maxmin": "16/23",
"code_range": "16/20",
"code_reduce": "15/33",
"code_reverse": "15/20",
"code_strings": "60/63"
},
"scoring": "generated function exec'd against held-out tests; expected values from verified reference"
},
"basic_code_screen": {
"ckpt": "ConeML/coneml-810m-alpha",
"by_type": {
"bash": "1/5",
"explain": "manual/10",
"script": "0/10",
"sql": "1/10"
}
},
"designated_refusal_probe": {
"ckpt": "ConeML/coneml-810m-alpha",
"refusals": "17/17",
"contrast_answers": "5/5",
"over_refusal": 0
},
"everyday_reasoning_screen": {
"n": 20,
"automatic_score": "6/20",
"manual_score": "11/20",
"adjudication": "manual adjudication; criterion: correct final choice, quantity, or agent regardless of phrasing"
},
"public_benchmarks": {
"model": {
"context": 8192,
"generation_budget": 256,
"gsm8k": {
"n": 200,
"correct": 12,
"acc": 0.06,
"truncation_rate": 0.01
},
"humaneval": {
"n": 164,
"passed": 5,
"pass@1": 0.0305,
"truncation_rate": 0.024
}
},
"pretrained_base_reference": {
"context": 8192,
"generation_budget": 256,
"gsm8k": {
"n": 200,
"correct": 7,
"acc": 0.035,
"truncation_rate": 0.315
},
"humaneval": {
"n": 164,
"passed": 0,
"pass@1": 0.0,
"truncation_rate": 0.665
}
}
},
"raw_completion_spot_check": {
"n": 12,
"seed": 188,
"temperature": 0.8,
"repetition_penalty": 1.15,
"repeated_4gram_rate": 0.0,
"distinct_2gram_rate": 0.9714,
"mean_words": 44.8
},
"conversation_probe": {
"probe": "conversation-v2",
"score": {
"correct": 8,
"n": 8,
"accuracy": 1.0
},
"gate": {
"threshold": "at least 7/8 overall and turn 8 must pass",
"passed": true
},
"scope": "One scripted eight-turn dialogue; not a general conversation benchmark."
}
},
"row_level_evidence": {
"public_subset": "representative-samples.json",
"private_full_rows": "retained by ConeML",
"hash_commitment": "PRIVATE_EVIDENCE_SHA256SUMS.txt"
}
}