{ "schema_version": 1, "model": "ConeML/coneml-810m-alpha", "scope": "Aggregate results from ConeML internal held-out instruments and the stated public-benchmark subsets. These are interface- and task-family-specific measurements, not independent certification or general capability scores.", "bf16": { "task_family_certification": { "ckpt": "ConeML/coneml-810m-alpha", "decoding": "greedy, rep_penalty 1.15, frame 'Question:/Answer:'", "by_category": { "add-1d": { "correct": 70, "n": 71, "acc": 0.9859 }, "add-2d-carry": { "correct": 266, "n": 500, "acc": 0.532 }, "comparison": { "correct": 363, "n": 500, "acc": 0.726 }, "missing-addend": { "correct": 358, "n": 500, "acc": 0.716 }, "missing-factor": { "correct": 89, "n": 500, "acc": 0.178 }, "mul-1d": { "correct": 44, "n": 45, "acc": 0.9778 }, "mul-2dx1d": { "correct": 487, "n": 500, "acc": 0.974 }, "sub-borrow": { "correct": 455, "n": 500, "acc": 0.91 }, "two-step": { "correct": 395, "n": 500, "acc": 0.79 }, "which-bigger": { "correct": 285, "n": 300, "acc": 0.95 } } }, "executed_single_function_writing": { "ckpt": "ConeML/coneml-810m-alpha", "passed": 207, "n": 300, "rate": 0.69, "by_family": { "code_arith": "18/24", "code_branch": "11/31", "code_count": "34/35", "code_dedup": "20/26", "code_filter": "2/25", "code_maxmin": "16/23", "code_range": "16/20", "code_reduce": "15/33", "code_reverse": "15/20", "code_strings": "60/63" }, "scoring": "generated function exec'd against held-out tests; expected values from verified reference" }, "basic_code_screen": { "ckpt": "ConeML/coneml-810m-alpha", "by_type": { "bash": "1/5", "explain": "manual/10", "script": "0/10", "sql": "1/10" } }, "designated_refusal_probe": { "ckpt": "ConeML/coneml-810m-alpha", "refusals": "17/17", "contrast_answers": "5/5", "over_refusal": 0 }, "everyday_reasoning_screen": { "n": 20, "automatic_score": "6/20", "manual_score": "11/20", "adjudication": "manual adjudication; criterion: correct final choice, quantity, or agent regardless of phrasing" }, "public_benchmarks": { "model": { "context": 8192, "generation_budget": 256, "gsm8k": { "n": 200, "correct": 12, "acc": 0.06, "truncation_rate": 0.01 }, "humaneval": { "n": 164, "passed": 5, "pass@1": 0.0305, "truncation_rate": 0.024 } }, "pretrained_base_reference": { "context": 8192, "generation_budget": 256, "gsm8k": { "n": 200, "correct": 7, "acc": 0.035, "truncation_rate": 0.315 }, "humaneval": { "n": 164, "passed": 0, "pass@1": 0.0, "truncation_rate": 0.665 } } }, "raw_completion_spot_check": { "n": 12, "seed": 188, "temperature": 0.8, "repetition_penalty": 1.15, "repeated_4gram_rate": 0.0, "distinct_2gram_rate": 0.9714, "mean_words": 44.8 }, "conversation_probe": { "probe": "conversation-v2", "score": { "correct": 8, "n": 8, "accuracy": 1.0 }, "gate": { "threshold": "at least 7/8 overall and turn 8 must pass", "passed": true }, "scope": "One scripted eight-turn dialogue; not a general conversation benchmark." } }, "row_level_evidence": { "public_subset": "representative-samples.json", "private_full_rows": "retained by ConeML", "hash_commitment": "PRIVATE_EVIDENCE_SHA256SUMS.txt" } }