wisp-coder-110m / evidence /results /e2_format_ablation.json
philipjohnbasile's picture
Publish audited Wisp Coder 110M release
818282c verified
Raw
History Blame Contribute Delete
3.31 kB
{
"arm_mean_acceptance": {
"run1_fim": 0.7848643666505813,
"run1_fim_shuf": 0.7643998256325721,
"run1_l2r": 0.779159582555294,
"run2_fim": 0.6985759252309799,
"run2_fim_shuf": 0.683045190423727,
"run2_l2r": 0.7791734331846237
},
"baseline_revision_evidence": {
"kind": "local_huggingface_cache_refs_captured_after_run1_corpus_build",
"limitation": "This is repository-revision evidence, not a per-document raw-corpus manifest for run 1. Source drift between the run 1 build and the captured cache refs cannot be ruled out."
},
"depth": 2,
"documents": 200,
"inputs": {
"ablation": {
"path": "out/run2-no-fim/acceptance.no-fim-ablation.json",
"sha256": "9a29fc1346b77fd3703e6b502f57276b9cf22aee904487a9ea2beac5e85a8681"
},
"baseline": {
"path": "out/run1/acceptance.trained.json",
"sha256": "53214c7cc098a3d29548563730855e16876ae28a615d61db62002dc170d81b0e"
},
"format_receipt": {
"path": "config/eval_format_ablation_receipt.json",
"sha256": "827ecc920fa6dd4cfe2922fde4d2289967c9e32759f368b466761302cf27b3c0"
},
"source_receipt": {
"path": "config/eval_holdout_receipt.json",
"sha256": "d24d4d8e41d75c91b2ff72a04f54a4519c15f1b362e70664cbfe1a12c736601d"
}
},
"instrument_version": 4,
"primary_endpoint": {
"ci95": [
0.07469401128590107,
0.0882960123345256
],
"contract": {
"bootstrap_samples": 4000,
"bootstrap_seed": 0,
"bootstrap_unit": "paired_target_document",
"depth": 2,
"metric": "(run1_fim_shuf_minus_l2r)_minus_(run2_fim_shuf_minus_l2r)",
"positive_when": "ci95_lower_gt_0"
},
"difference_in_differences": 0.08136848583817483,
"run1_fim_shuf_minus_l2r": -0.014759756922721912,
"run2_fim_shuf_minus_l2r": -0.09612824276089671,
"verdict": "POSITIVE"
},
"publication_ready": true,
"research_outcomes_are_not_release_gates": true,
"runtime_code_evidence": {
"kind": "matched_sampler_reset_with_post_build_shard_attestation",
"limitation": "Run 1 did not record source-file hashes or sampler state in its pre-fix checkpoints. A post-build receipt now hashes all 52 current shard files and binds deterministic visible-grammar normalization, but it does not prove that the bytes were unchanged since training began or recover exact original units. The step-300 process recovery reset the legacy sampler, causing 78,643,200 scheduled token positions, 1.57 percent of the training budget, to replay earlier random windows. sampler_reset_steps [300] matches that reset timing in both arms. Later checkpoints preserve exact sampler RNG state. Exact sampled-example equivalence still cannot be proven."
},
"schema_version": 1,
"secondary_endpoint": {
"ci95": [
-0.0025777602791786196,
0.012247835688292982
],
"contract": {
"bootstrap_samples": 4000,
"bootstrap_seed": 0,
"bootstrap_unit": "paired_target_document",
"depth": 2,
"metric": "(run1_fim_minus_fim_shuf)_minus_(run2_fim_minus_fim_shuf)"
},
"difference_in_differences": 0.004933806210756302,
"run1_fim_minus_fim_shuf": 0.020464541018009208,
"run2_fim_minus_fim_shuf": 0.015530734807252866,
"verdict": "NULL: the interval includes zero"
}
}