coneml-810m-alpha-arithmetic / eval /peer-comparison-summary.json
RandomMountainMan's picture
Clarify 810M family naming
352376e verified
Raw
History Blame Contribute Delete
7.97 kB
{
"schema_version": 1,
"date": "2026-07-30",
"scope": "Interface-specific diagnostic screen, not a neutral ranking of general model capability. Each instruct model used its native instruction interface. Arithmetic and function-writing used matched short-answer generation budgets.",
"decoding": {
"strategy": "greedy",
"repetition_penalty": 1.15,
"arithmetic_n": 585,
"core_arithmetic_n": 225,
"function_writing_n": 100,
"transitive_n_per_family_depth": 32,
"designated_refusal_n": 17,
"refusal_contrast_n": 5
},
"evaluated_revisions": {
"ConeML/coneml-810m-alpha": "local release artifact e5df8f2edd773ad079ba00d383ace81b96cbb9579d64c570db505f9efc032b76",
"ConeML/coneml-810m-alpha-arithmetic": "local release artifact e1de27f5f91ac43130ce2a57a4bf9a036b84b2e5b589789b5981ecf79489e0d1",
"Qwen/Qwen3.5-0.8B": "2fc06364715b967f1860aea9cf38778875588b17",
"Qwen/Qwen3-0.6B": "c1899de289a04d12100db370d81485cdf75e47ca",
"unsloth/Llama-3.2-1B-Instruct": "5a8abab4a5d6f164389b1079fb721cfab8d7126c",
"TinyLlama/TinyLlama-1.1B-Chat-v1.0": "fe8a4ea1ffedaf415f4da2f062534de366a451e6",
"HuggingFaceTB/SmolLM2-1.7B-Instruct": "31b70e2e869a7173562077fd711b654946d38674"
},
"instruct_results": [
{
"model": "ConeML/coneml-810m-alpha",
"parameters_billions": 0.8101,
"arithmetic_mixed": {"correct": 421, "n": 585, "accuracy": 0.7197},
"arithmetic_core_four_lanes": {"correct": 177, "n": 225, "accuracy": 0.7867},
"executed_functions": {"correct": 83, "n": 100, "accuracy": 0.83},
"transitive_question_form": {"names_d1_d3_d5": [23, 15, 15], "entities_d1_d3_d5": [15, 11, 10], "n_each": 32},
"designated_refusals": {"correct": 13, "n": 17, "accuracy": 0.7647},
"over_refusals": {"count": 0, "n": 5}
},
{
"model": "ConeML/coneml-810m-alpha-arithmetic",
"parameters_billions": 0.8101,
"arithmetic_mixed": {"correct": 442, "n": 585, "accuracy": 0.7556},
"arithmetic_core_four_lanes": {"correct": 221, "n": 225, "accuracy": 0.9822},
"executed_functions": {"correct": 35, "n": 100, "accuracy": 0.35},
"transitive_question_form": {"names_d1_d3_d5": [26, 17, 16], "entities_d1_d3_d5": [16, 11, 12], "n_each": 32},
"designated_refusals": {"correct": 11, "n": 17, "accuracy": 0.6471},
"over_refusals": {"count": 0, "n": 5}
},
{
"model": "Qwen/Qwen3.5-0.8B",
"parameters_billions": 0.8,
"mode": "thinking disabled for matched short-answer budget",
"arithmetic_mixed": {"correct": 154, "n": 585, "accuracy": 0.2632},
"arithmetic_core_four_lanes": {"correct": 150, "n": 225, "accuracy": 0.6667},
"executed_functions": {"correct": 93, "n": 100, "accuracy": 0.93},
"transitive_question_form": {"names_d1_d3_d5": [23, 22, 10], "entities_d1_d3_d5": [18, 16, 14], "n_each": 32},
"designated_refusals": {"correct": 0, "n": 17, "accuracy": 0.0},
"over_refusals": {"count": 0, "n": 5}
},
{
"model": "Qwen/Qwen3-0.6B",
"parameters_billions": 0.6,
"arithmetic_mixed": {"correct": 223, "n": 585, "accuracy": 0.3812},
"arithmetic_core_four_lanes": {"correct": 208, "n": 225, "accuracy": 0.9244},
"executed_functions": {"correct": 98, "n": 100, "accuracy": 0.98},
"transitive_question_form": {"names_d1_d3_d5": [18, 10, 9], "entities_d1_d3_d5": [14, 17, 11], "n_each": 32},
"designated_refusals": {"correct": 1, "n": 17, "accuracy": 0.0588},
"over_refusals": {"count": 0, "n": 5}
},
{
"model": "unsloth/Llama-3.2-1B-Instruct",
"parameters_billions": 1.24,
"arithmetic_mixed": {"correct": 343, "n": 585, "accuracy": 0.5863},
"arithmetic_core_four_lanes": {"correct": 221, "n": 225, "accuracy": 0.9822},
"executed_functions": {"correct": 79, "n": 100, "accuracy": 0.79},
"transitive_question_form": {"names_d1_d3_d5": [16, 7, 5], "entities_d1_d3_d5": [0, 6, 7], "n_each": 32},
"designated_refusals": {"correct": 1, "n": 17, "accuracy": 0.0588},
"over_refusals": {"count": 0, "n": 5}
},
{
"model": "TinyLlama/TinyLlama-1.1B-Chat-v1.0",
"parameters_billions": 1.1,
"arithmetic_mixed": {"correct": 113, "n": 585, "accuracy": 0.1932},
"arithmetic_core_four_lanes": {"correct": 92, "n": 225, "accuracy": 0.4089},
"executed_functions": {"correct": 51, "n": 100, "accuracy": 0.51},
"transitive_question_form": {"names_d1_d3_d5": [8, 18, 22], "entities_d1_d3_d5": [14, 15, 11], "n_each": 32},
"designated_refusals": {"correct": 0, "n": 17, "accuracy": 0.0},
"over_refusals": {"count": 0, "n": 5}
},
{
"model": "HuggingFaceTB/SmolLM2-1.7B-Instruct",
"parameters_billions": 1.7,
"arithmetic_mixed": {"correct": 380, "n": 585, "accuracy": 0.6496},
"arithmetic_core_four_lanes": {"correct": 221, "n": 225, "accuracy": 0.9822},
"executed_functions": {"correct": 96, "n": 100, "accuracy": 0.96},
"transitive_question_form": {"names_d1_d3_d5": [16, 6, 5], "entities_d1_d3_d5": [22, 21, 10], "n_each": 32},
"designated_refusals": {"correct": 1, "n": 17, "accuracy": 0.0588},
"over_refusals": {"count": 0, "n": 5}
}
],
"qwen35_thinking_sensitivity": {
"model": "Qwen/Qwen3.5-0.8B",
"mode": "thinking enabled",
"arithmetic_mixed": {"correct": 451, "n": 585, "accuracy": 0.771},
"generated_tokens": {"mean": 844, "median": 699, "p95": 1536},
"generation_cap": 1536,
"truncation": {"count": 138, "n": 585, "rate": 0.2359},
"wall_seconds_per_item_mean": 2.43,
"correct_per_1000_generated_tokens": 0.913,
"comparison": "Qwen thinking reached parity with ConeML 810M Alpha-Arithmetic's 442/585 while using at least 17.6 times the generated-token budget per item. The recorded wall-time ratio was approximately 24 times. Runtime measurements are hardware- and implementation-specific."
},
"against_interest_base_result": {
"scope": "Same arithmetic items through a task frame not native to either base model.",
"ConeML_base_188": {"correct": 156, "n": 585, "accuracy": 0.2667},
"Qwen3.5_0.8B_Base": {"correct": 492, "n": 585, "accuracy": 0.841},
"interpretation": "The peer base result prevents an absolute-superiority interpretation and illustrates sensitivity to training exposure and interface."
},
"training_resource_context": {
"ConeML": {
"pretraining_token_positions": 12320000000,
"tokens_per_parameter": 15.2,
"estimated_training_flops_6ND": 5.99e19,
"pretraining_wall_time_days": 11,
"hardware": "one NVIDIA RTX 5090",
"power_assumption_kw_average_wall": 0.70,
"estimated_pretraining_energy_kwh": 184.8,
"assumed_swiss_residential_tariff_chf_per_kwh": [0.14, 0.30],
"estimated_marginal_pretraining_electricity_chf": [25.87, 55.44],
"exclusions": ["hardware", "depreciation", "labor", "SFT", "evaluation", "conversion", "grid-carbon estimate"]
},
"peer_compute_context": [
{"model": "TinyLlama 1.1B", "published_pretraining_tokens": 3000000000000, "approx_tokens_per_parameter": 2727, "approx_flops_vs_coneml": 330},
{"model": "Llama 3.2 1B", "published_pretraining_tokens": 9000000000000, "approx_tokens_per_parameter": 7258, "approx_flops_vs_coneml": 1118, "note": "plus distillation"},
{"model": "SmolLM2 1.7B", "published_pretraining_tokens": 11000000000000, "approx_tokens_per_parameter": 6471, "approx_flops_vs_coneml": 1873},
{"model": "Qwen3 0.6B", "published_pretraining_tokens": 36000000000000, "approx_tokens_per_parameter": 60000, "approx_flops_vs_coneml": 2164},
{"model": "Qwen3.5 0.8B", "published_pretraining_tokens": null, "approx_tokens_per_parameter": null, "approx_flops_vs_coneml": null}
]
},
"evidence": {
"public_interpretation": "PEER_COMPARISON.md",
"private_rows": "retained by ConeML",
"hash_commitment": "PEER_EVIDENCE_SHA256SUMS.txt"
}
}