{ "schema_version": 1, "date": "2026-07-30", "scope": "Interface-specific diagnostic screen, not a neutral ranking of general model capability. Each instruct model used its native instruction interface. Arithmetic and function-writing used matched short-answer generation budgets.", "decoding": { "strategy": "greedy", "repetition_penalty": 1.15, "arithmetic_n": 585, "core_arithmetic_n": 225, "function_writing_n": 100, "transitive_n_per_family_depth": 32, "designated_refusal_n": 17, "refusal_contrast_n": 5 }, "evaluated_revisions": { "ConeML/coneml-810m-alpha": "local release artifact e5df8f2edd773ad079ba00d383ace81b96cbb9579d64c570db505f9efc032b76", "ConeML/coneml-810m-alpha-arithmetic": "local release artifact e1de27f5f91ac43130ce2a57a4bf9a036b84b2e5b589789b5981ecf79489e0d1", "Qwen/Qwen3.5-0.8B": "2fc06364715b967f1860aea9cf38778875588b17", "Qwen/Qwen3-0.6B": "c1899de289a04d12100db370d81485cdf75e47ca", "unsloth/Llama-3.2-1B-Instruct": "5a8abab4a5d6f164389b1079fb721cfab8d7126c", "TinyLlama/TinyLlama-1.1B-Chat-v1.0": "fe8a4ea1ffedaf415f4da2f062534de366a451e6", "HuggingFaceTB/SmolLM2-1.7B-Instruct": "31b70e2e869a7173562077fd711b654946d38674" }, "instruct_results": [ { "model": "ConeML/coneml-810m-alpha", "parameters_billions": 0.8101, "arithmetic_mixed": {"correct": 421, "n": 585, "accuracy": 0.7197}, "arithmetic_core_four_lanes": {"correct": 177, "n": 225, "accuracy": 0.7867}, "executed_functions": {"correct": 83, "n": 100, "accuracy": 0.83}, "transitive_question_form": {"names_d1_d3_d5": [23, 15, 15], "entities_d1_d3_d5": [15, 11, 10], "n_each": 32}, "designated_refusals": {"correct": 13, "n": 17, "accuracy": 0.7647}, "over_refusals": {"count": 0, "n": 5} }, { "model": "ConeML/coneml-810m-alpha-arithmetic", "parameters_billions": 0.8101, "arithmetic_mixed": {"correct": 442, "n": 585, "accuracy": 0.7556}, "arithmetic_core_four_lanes": {"correct": 221, "n": 225, "accuracy": 0.9822}, "executed_functions": {"correct": 35, "n": 100, "accuracy": 0.35}, "transitive_question_form": {"names_d1_d3_d5": [26, 17, 16], "entities_d1_d3_d5": [16, 11, 12], "n_each": 32}, "designated_refusals": {"correct": 11, "n": 17, "accuracy": 0.6471}, "over_refusals": {"count": 0, "n": 5} }, { "model": "Qwen/Qwen3.5-0.8B", "parameters_billions": 0.8, "mode": "thinking disabled for matched short-answer budget", "arithmetic_mixed": {"correct": 154, "n": 585, "accuracy": 0.2632}, "arithmetic_core_four_lanes": {"correct": 150, "n": 225, "accuracy": 0.6667}, "executed_functions": {"correct": 93, "n": 100, "accuracy": 0.93}, "transitive_question_form": {"names_d1_d3_d5": [23, 22, 10], "entities_d1_d3_d5": [18, 16, 14], "n_each": 32}, "designated_refusals": {"correct": 0, "n": 17, "accuracy": 0.0}, "over_refusals": {"count": 0, "n": 5} }, { "model": "Qwen/Qwen3-0.6B", "parameters_billions": 0.6, "arithmetic_mixed": {"correct": 223, "n": 585, "accuracy": 0.3812}, "arithmetic_core_four_lanes": {"correct": 208, "n": 225, "accuracy": 0.9244}, "executed_functions": {"correct": 98, "n": 100, "accuracy": 0.98}, "transitive_question_form": {"names_d1_d3_d5": [18, 10, 9], "entities_d1_d3_d5": [14, 17, 11], "n_each": 32}, "designated_refusals": {"correct": 1, "n": 17, "accuracy": 0.0588}, "over_refusals": {"count": 0, "n": 5} }, { "model": "unsloth/Llama-3.2-1B-Instruct", "parameters_billions": 1.24, "arithmetic_mixed": {"correct": 343, "n": 585, "accuracy": 0.5863}, "arithmetic_core_four_lanes": {"correct": 221, "n": 225, "accuracy": 0.9822}, "executed_functions": {"correct": 79, "n": 100, "accuracy": 0.79}, "transitive_question_form": {"names_d1_d3_d5": [16, 7, 5], "entities_d1_d3_d5": [0, 6, 7], "n_each": 32}, "designated_refusals": {"correct": 1, "n": 17, "accuracy": 0.0588}, "over_refusals": {"count": 0, "n": 5} }, { "model": "TinyLlama/TinyLlama-1.1B-Chat-v1.0", "parameters_billions": 1.1, "arithmetic_mixed": {"correct": 113, "n": 585, "accuracy": 0.1932}, "arithmetic_core_four_lanes": {"correct": 92, "n": 225, "accuracy": 0.4089}, "executed_functions": {"correct": 51, "n": 100, "accuracy": 0.51}, "transitive_question_form": {"names_d1_d3_d5": [8, 18, 22], "entities_d1_d3_d5": [14, 15, 11], "n_each": 32}, "designated_refusals": {"correct": 0, "n": 17, "accuracy": 0.0}, "over_refusals": {"count": 0, "n": 5} }, { "model": "HuggingFaceTB/SmolLM2-1.7B-Instruct", "parameters_billions": 1.7, "arithmetic_mixed": {"correct": 380, "n": 585, "accuracy": 0.6496}, "arithmetic_core_four_lanes": {"correct": 221, "n": 225, "accuracy": 0.9822}, "executed_functions": {"correct": 96, "n": 100, "accuracy": 0.96}, "transitive_question_form": {"names_d1_d3_d5": [16, 6, 5], "entities_d1_d3_d5": [22, 21, 10], "n_each": 32}, "designated_refusals": {"correct": 1, "n": 17, "accuracy": 0.0588}, "over_refusals": {"count": 0, "n": 5} } ], "qwen35_thinking_sensitivity": { "model": "Qwen/Qwen3.5-0.8B", "mode": "thinking enabled", "arithmetic_mixed": {"correct": 451, "n": 585, "accuracy": 0.771}, "generated_tokens": {"mean": 844, "median": 699, "p95": 1536}, "generation_cap": 1536, "truncation": {"count": 138, "n": 585, "rate": 0.2359}, "wall_seconds_per_item_mean": 2.43, "correct_per_1000_generated_tokens": 0.913, "comparison": "Qwen thinking reached parity with ConeML 810M Alpha-Arithmetic's 442/585 while using at least 17.6 times the generated-token budget per item. The recorded wall-time ratio was approximately 24 times. Runtime measurements are hardware- and implementation-specific." }, "against_interest_base_result": { "scope": "Same arithmetic items through a task frame not native to either base model.", "ConeML_base_188": {"correct": 156, "n": 585, "accuracy": 0.2667}, "Qwen3.5_0.8B_Base": {"correct": 492, "n": 585, "accuracy": 0.841}, "interpretation": "The peer base result prevents an absolute-superiority interpretation and illustrates sensitivity to training exposure and interface." }, "training_resource_context": { "ConeML": { "pretraining_token_positions": 12320000000, "tokens_per_parameter": 15.2, "estimated_training_flops_6ND": 5.99e19, "pretraining_wall_time_days": 11, "hardware": "one NVIDIA RTX 5090", "power_assumption_kw_average_wall": 0.70, "estimated_pretraining_energy_kwh": 184.8, "assumed_swiss_residential_tariff_chf_per_kwh": [0.14, 0.30], "estimated_marginal_pretraining_electricity_chf": [25.87, 55.44], "exclusions": ["hardware", "depreciation", "labor", "SFT", "evaluation", "conversion", "grid-carbon estimate"] }, "peer_compute_context": [ {"model": "TinyLlama 1.1B", "published_pretraining_tokens": 3000000000000, "approx_tokens_per_parameter": 2727, "approx_flops_vs_coneml": 330}, {"model": "Llama 3.2 1B", "published_pretraining_tokens": 9000000000000, "approx_tokens_per_parameter": 7258, "approx_flops_vs_coneml": 1118, "note": "plus distillation"}, {"model": "SmolLM2 1.7B", "published_pretraining_tokens": 11000000000000, "approx_tokens_per_parameter": 6471, "approx_flops_vs_coneml": 1873}, {"model": "Qwen3 0.6B", "published_pretraining_tokens": 36000000000000, "approx_tokens_per_parameter": 60000, "approx_flops_vs_coneml": 2164}, {"model": "Qwen3.5 0.8B", "published_pretraining_tokens": null, "approx_tokens_per_parameter": null, "approx_flops_vs_coneml": null} ] }, "evidence": { "public_interpretation": "PEER_COMPARISON.md", "private_rows": "retained by ConeML", "hash_commitment": "PEER_EVIDENCE_SHA256SUMS.txt" } }