{ "date": "2026-08-25", "type": "controlled regression samples; not a full official leaderboard run", "seed": 20260819, "decoding": { "method": "greedy", "budgets": { "gsm8k": 256, "math500": 384, "arc_challenge": 128, "folio": 160 } }, "models": { "budgie": { "repo": "Asilarkness/Budgie-500m", "checkpoint": "verified-math-a025", "source_repo": "Asilarkness/testgeniy", "source_path": "candidates/budgie-alignment-v2/verified-math-a025", "source_revision": "79aec5157c10e3f7e082823a6e9150482d740a91" }, "previous_public": { "repo": "Asilarkness/Budgie-500m", "revision": "39c877f07ddc80aa975809a4d87c848372316caf" }, "qwen3": { "repo": "Qwen/Qwen3-0.6B", "revision": "c1899de289a04d12100db370d81485cdf75e47ca", "thinking": false } }, "datasets": { "GSM8K": "openai/gsm8k main test, fixed n=30", "MATH-500": "HuggingFaceH4/MATH-500 test, fixed n=15", "ARC-Challenge": "allenai/ai2_arc ARC-Challenge test, fixed n=30", "FOLIO": "tasksource/folio validation, fixed n=30", "HelpSteer2-pairwise": "nvidia/HelpSteer2 validation, fixed n=200" }, "previous_root_results": { "GSM8K": { "correct": 4, "total": 30, "score": 13.333333333333334 }, "MATH-500": { "correct": 2, "total": 15, "score": 13.333333333333334 }, "ARC-Challenge": { "correct": 8, "total": 30, "score": 26.666666666666668 }, "FOLIO": { "correct": 11, "total": 30, "score": 36.666666666666664 }, "HelpSteer2-pairwise": { "correct": 96, "total": 200, "score": 48.0 }, "Composite": { "score": 27.6 } }, "results": { "budgie": { "GSM8K": { "correct": 5, "total": 30, "score": 16.666666666666664 }, "MATH-500": { "correct": 3, "total": 15, "score": 20.0 }, "ARC-Challenge": { "correct": 11, "total": 30, "score": 36.666666666666664 }, "FOLIO": { "correct": 13, "total": 30, "score": 43.333333333333336 }, "HelpSteer2-pairwise": { "correct": 97, "total": 200, "score": 48.5 }, "Composite": { "score": 33.03333333333333 } }, "qwen3": { "GSM8K": { "correct": 14, "total": 30, "score": 46.666666666666664 }, "MATH-500": { "correct": 2, "total": 15, "score": 13.333333333333334 }, "ARC-Challenge": { "correct": 19, "total": 30, "score": 63.333333333333336 }, "FOLIO": { "correct": 13, "total": 30, "score": 43.333333333333336 }, "HelpSteer2-pairwise": { "correct": 100, "total": 200, "score": 50.0 }, "Composite": { "score": 43.33333333333333 } } }, "expanded_disjoint_gate": { "seed": 20260908, "split": "disjoint expanded gate; excludes every row in the original controlled gate", "results": { "GSM8K": { "total": 80, "previous_correct": 9, "previous_score": 11.25, "current_correct": 24, "current_score": 30.0 }, "MATH-500": { "total": 50, "previous_correct": 10, "previous_score": 20.0, "current_correct": 10, "current_score": 20.0 }, "ARC-Challenge": { "total": 80, "previous_correct": 16, "previous_score": 20.0, "current_correct": 25, "current_score": 31.25 }, "FOLIO": { "total": 100, "previous_correct": 31, "previous_score": 31.0, "current_correct": 36, "current_score": 36.0 }, "HelpSteer2-pairwise": { "total": 373, "previous_correct": 166, "previous_score": 44.50402144772118, "current_correct": 169, "current_score": 45.30831099195711 }, "Composite": { "previous_score": 25.35080428954424, "current_score": 32.51166219839142, "delta": 7.160857908847185 } }, "paired_bootstrap": { "samples": 30000, "current_composite_95": [ 28.34289544235925, 36.82319034852547 ], "delta_95": [ 2.5861930294906132, 11.725335120643432 ], "probability_delta_positive": 99.86333333333333 } }, "training_disclosure": { "no_fixed_test_or_validation_row_used_for_training": true, "decontamination": "normalized exact matching, long containment checks, deduplication, and shared 12-token-window filtering against fixed development/final gates", "legacy_training_split_sources": [ "GSM8K train", "ARC-Challenge train", "Hendrycks MATH train", "ProofWriter train" ], "other_sources": [ "NuminaMath-CoT", "Bespoke-Stratos-17k", "SmolTalk / Smol-SmolTalk", "Qwen3.8-27B synthetic and distilled answers", "random verified mathematics" ], "note": "This retained checkpoint predates the stricter benchmark-evaluation-only policy now used for future work. Training-split use is disclosed here; no reported held-out row was used for training." }, "full_split_comparison": { "date": "2026-08-26", "type": "complete-split controlled comparison; not an official leaderboard submission", "models": { "old_public": { "repo": "Asilarkness/Budgie-500m", "revision": "39c877f07ddc80aa975809a4d87c848372316caf", "weight_sha256": "10845d227ea7ea50d14757df426e5265820c6fe1d48741b0c38b4fc23bd2381e" }, "current_leader": { "repo": "Asilarkness/Budgie-500m", "weights_revision": "34f0d09538dcca578ea672ba717e8e9f2672021f", "checkpoint": "ba2_vmath_a025 / verified-math-a025", "weight_sha256": "7404de2c08df968f41800d1362704a0ad4106ea0116cd0cd4b1709612a2a66be" } }, "datasets": { "gsm8k": { "repo": "openai/gsm8k", "config": "main", "split": "test", "n": 1319 }, "math500": { "repo": "HuggingFaceH4/MATH-500", "split": "test", "n": 500 }, "arc_challenge": { "repo": "allenai/ai2_arc", "config": "ARC-Challenge", "split": "test", "n": 1172 }, "folio": { "repo": "tasksource/folio", "split": "validation", "n": 203 }, "helpsteer2": { "repo": "nvidia/HelpSteer2", "split": "validation", "paired_non_tied_rows": 373, "note": "Reused complete 373-pair evaluation already computed for these exact revisions." } }, "protocol": { "prompts": "same user-only task prompts for both models", "decoding": "greedy", "budgets": { "gsm8k": 256, "math500": 384, "arc_challenge": 128, "folio": 160 }, "scoring": { "gsm8k": "final numeric answer exact match", "math500": "math_verify symbolic equivalence against reference solution, main thread", "arc_challenge": "final option label; handles letter and numeric label variants", "folio": "final True/False/Uncertain exact label", "helpsteer2": "lower assistant-answer NLL on human-preferred response" }, "batching": "Only prompts with identical tokenized input length were batched. No input padding. Each row was truncated manually at its first EOS. A two-prompt audit exactly matched sequential generation.", "invalid_attempt": "An earlier unequal-length padded batch was detected as corrupt and discarded; none of those outputs enter this report." }, "runtime": { "torch": "2.11.0+cu130", "transformers": "5.15.1", "datasets": "5.0.0" }, "results": { "tasks": { "gsm8k": { "n": 1319, "old_correct": 191, "current_correct": 245, "old_score": 0.14480667172100076, "current_score": 0.18574677786201668, "delta": 0.04094010614101592, "improved": 143, "worsened": 89, "paired_binomial_p": 0.00047624975354938266 }, "math500": { "n": 500, "old_correct": 62, "current_correct": 60, "old_score": 0.124, "current_score": 0.12, "delta": -0.004, "improved": 24, "worsened": 26, "paired_binomial_p": 0.887724827340783 }, "arc_challenge": { "n": 1172, "old_correct": 291, "current_correct": 323, "old_score": 0.24829351535836178, "current_score": 0.27559726962457337, "delta": 0.027303754266211604, "improved": 246, "worsened": 214, "paired_binomial_p": 0.14827050225768382 }, "folio": { "n": 203, "old_correct": 63, "current_correct": 72, "old_score": 0.3103448275862069, "current_score": 0.35467980295566504, "delta": 0.04433497536945813, "improved": 25, "worsened": 16, "paired_binomial_p": 0.21102359760880063 }, "helpsteer2": { "n": 373, "old_correct": 166, "current_correct": 169, "old_score": 0.4450402144772118, "current_score": 0.45308310991957107, "delta": 0.00804289544235925, "improved": 9, "worsened": 6, "paired_binomial_p": 0.6072387695312499 } }, "reasoning_macro_old": 0.20686125366639235, "reasoning_macro_current": 0.23400596261056378, "reasoning_delta": 0.027144708944171425, "five_metric_composite_old": 0.25449704582855626, "five_metric_composite_current": 0.2778213920723652, "five_metric_delta": 0.02332434624380897, "paired_bootstrap_delta_95": [ 0.00678765357806182, 0.03969201450574924 ], "probability_delta_positive": 0.9978666666666667, "bootstrap_samples": 30000 }, "artifact": "https://huggingface.co/Asilarkness/testgeniy/tree/main/candidates/budgie-alignment-v2/fullbench-old-vs-vmath25-v1" } }