occamy-1.0-FP8 / EXPANDED-VALIDATION.json
Eang's picture
Release Occamy 1.0 FP8
ac48ca6
Raw
History Blame Contribute Delete
5.02 kB
{
"scope": "192 paired cases per variant; expanded sampled regression evaluation, not full-dataset leaderboard results",
"baseline": {
"arc": {
"passed": 59,
"total": 64,
"rate": 0.921875,
"wilson95": [
0.8298024712135222,
0.9661693219045645
],
"errors": 0,
"truncated": 0
},
"code": {
"passed": 12,
"total": 12,
"rate": 1.0,
"wilson95": [
0.7574992425007574,
1
],
"errors": 0,
"truncated": 0
},
"gsm8k": {
"passed": 63,
"total": 64,
"rate": 0.984375,
"wilson95": [
0.9166570109970937,
0.9972365292495986
],
"errors": 0,
"truncated": 0
},
"json": {
"passed": 20,
"total": 20,
"rate": 1.0,
"wilson95": [
0.8388698745050667,
1
],
"errors": 0,
"truncated": 0
},
"long_context": {
"passed": 6,
"total": 6,
"rate": 1.0,
"wilson95": [
0.6096569663469354,
0.9999999999999999
],
"errors": 0,
"truncated": 0
},
"multiturn": {
"passed": 8,
"total": 8,
"rate": 1.0,
"wilson95": [
0.6755843804891231,
1
],
"errors": 0,
"truncated": 0
},
"tool": {
"passed": 6,
"total": 6,
"rate": 1.0,
"wilson95": [
0.6096569663469354,
0.9999999999999999
],
"errors": 0,
"truncated": 0
},
"vision": {
"passed": 11,
"total": 12,
"rate": 0.9166666666666666,
"wilson95": [
0.6461140782014047,
0.9851352905492264
],
"errors": 0,
"truncated": 0
}
},
"fp8": {
"arc": {
"passed": 59,
"total": 64,
"rate": 0.921875,
"wilson95": [
0.8298024712135222,
0.9661693219045645
],
"errors": 0,
"truncated": 0
},
"code": {
"passed": 12,
"total": 12,
"rate": 1.0,
"wilson95": [
0.7574992425007574,
1
],
"errors": 0,
"truncated": 0
},
"gsm8k": {
"passed": 62,
"total": 64,
"rate": 0.96875,
"wilson95": [
0.8930250611770575,
0.9913880422874833
],
"errors": 0,
"truncated": 0
},
"json": {
"passed": 20,
"total": 20,
"rate": 1.0,
"wilson95": [
0.8388698745050667,
1
],
"errors": 0,
"truncated": 0
},
"long_context": {
"passed": 6,
"total": 6,
"rate": 1.0,
"wilson95": [
0.6096569663469354,
0.9999999999999999
],
"errors": 0,
"truncated": 0
},
"multiturn": {
"passed": 8,
"total": 8,
"rate": 1.0,
"wilson95": [
0.6755843804891231,
1
],
"errors": 0,
"truncated": 0
},
"tool": {
"passed": 6,
"total": 6,
"rate": 1.0,
"wilson95": [
0.6096569663469354,
0.9999999999999999
],
"errors": 0,
"truncated": 0
},
"vision": {
"passed": 10,
"total": 12,
"rate": 0.8333333333333334,
"wilson95": [
0.5519636426153274,
0.9530358523851776
],
"errors": 0,
"truncated": 0
}
},
"paired_regressions_first_run": [
"gsm8k-017",
"vision-004"
],
"serial_failure_confirmation": {
"sglang-bf16": {
"gsm8k-017": {
"passes": 3,
"attempts": 3
},
"vision-004": {
"passes": 0,
"attempts": 3
}
},
"sglang-fp8": {
"gsm8k-017": {
"passes": 0,
"attempts": 3
},
"vision-004": {
"passes": 0,
"attempts": 3
}
}
},
"nll": {
"dataset": "Salesforce/wikitext",
"records": 16,
"tokens_each": 4080,
"bf16_nll": 2.206196128492876,
"fp8_nll": 2.213649180375206,
"relative_nll_change": 0.003378236316379235,
"perplexity_ratio": 1.0074808950026004,
"finite": true,
"matched_tokens": true,
"predeclared_nll_limit": 0.05,
"nll_gate_pass": true
},
"errors": 0,
"truncated_requests": 0,
"confirmed_regression": "gsm8k-017: BF16 correct 3/3, FP8 incorrect 3/3 in serial repeats. Quantization is not lossless.",
"ocr_caveat": "vision-004 differs in the four-concurrent-request run; both BF16 and FP8 fail 3/3 serial repeats, so it is not established as a stable FP8-only regression.",
"older_smoke_json": "A separate earlier four-JSON smoke subset passed 3/4 for both models; one response included Markdown. Expanded 20/20 does not erase that failure.",
"mtp_supported": false,
"limits": [
"No broad natural-image, audio/video, maximum-context, Unsloth or training validation.",
"Shared-resource timings are not benchmark-grade throughput measurements.",
"Dataset samples and synthetic fixtures do not establish broad quality parity."
]
}