soma-cot-compression / acon_axis_report.json
XXMiner's picture
v99: ACON directional scopes map the frontier; doc-unique proven a strict subset of forward-unique; fwd/bwd union covers 20/25 lost terms; 501 tests
97490d7 verified
Raw
History Blame Contribute Delete
6.97 kB
{
"axis": "ACON arXiv:2510.00615 - keep what later steps need - applied to the routing problem I had failed at from four directions",
"arxiv_citations": [
{
"title": "ACON: Optimizing Context Compression for Long-horizon LLM Agents",
"id": "arXiv:2510.00615",
"venue": "ICML 2026",
"date": "2025-10-01"
},
{
"title": "Distilling LLM Agent into Small Models with Retrieval and Code Tools",
"id": "arXiv:2505.17612"
},
{
"title": "Structured Agent Distillation for Large Language Models",
"id": "arXiv:2505.13820"
}
],
"acon_gap": "ACON fits its compressor from outcome feedback (compare success with full vs compressed context). I have no run outcomes - no OpenRouter key - so I took only its structural principle: directionality",
"challenge_level_rates_useless_again": {
"results": {
"fwd": -0.0081,
"bwd": -0.0017,
"doc": -0.0092
},
"reading": "all three separations are near zero and NEGATIVE at the challenge level; I ran this first out of habit and correctly refused to conclude from it, because v95 established that a challenge-level rate is the wrong instrument for a span-level rule. Second time in three turns it pointed the opposite way from term coverage"
},
"span_lattice": {
"results": {
"fwd": 452,
"bwd": 509,
"doc": 127,
"doc_subset_of_fwd": true,
"doc_minus_fwd": 0,
"doc_subset_of_bwd": true,
"fwd_and_bwd": 138,
"fwd_or_bwd": 823,
"fwd_only": 314,
"bwd_only": 371
},
"finding": "document-unique is a STRICT SUBSET of both forward-unique and backward-unique (0 spans outside either) - logically necessary since \"appears nowhere else\" implies \"appears nowhere later\"",
"how_i_found_it": "I tried to write a fwd-OR-doc union variant and saw the disjunction collapse; the subset relation was only visible once I wrote the predicate down",
"reframing": "this corrects how I described the forward scope last turn. It is not a DIFFERENT feature that beat mine, it is a strict RELAXATION of mine. My document-wide rule was over-restrictive at 127 spans when the useful set was 452. The gain came from loosening a too-tight condition, not from directionality per se",
"non_vacuous_pair": "fwd and bwd share only 138 of 823 spans, so they are genuinely different and their union is the one combination worth building"
},
"term_lattice": {
"results": {
"total_lost_terms": 25,
"fwd": 16,
"bwd": 15,
"doc": 8,
"fwd_and_bwd": 11,
"fwd_or_bwd": 20,
"fwd_only": 5,
"bwd_only": 4,
"doc_subset_of_fwd_terms": true,
"uncovered_by_either": 5
},
"finding": "fwd covers 16 of 25 lost terms, bwd 15, sharing only 11 - so 5 are forward-only and 4 backward-only, and the union reaches 20 of 25 (80%)",
"why_it_matters": "the first combination on this axis with headroom over its parts; only 5 lost terms are unreachable by either scope"
},
"frontier": {
"points": [
{
"variant": "shipped",
"spans": 0,
"terms_covered": 0,
"char_pct": 78.136,
"term_10dp": 0.935724891
},
{
"variant": "doc",
"spans": 127,
"terms_covered": 8,
"char_pct": 80.896,
"term_10dp": 0.9344322611
},
{
"variant": "forward",
"spans": 452,
"terms_covered": 16,
"char_pct": 80.656,
"term_10dp": 0.9348915453
},
{
"variant": "combined",
"spans": 823,
"terms_covered": 20,
"char_pct": 80.274,
"term_10dp": 0.9351396283
}
],
"monotone": "every step up in coverage buys retention and costs compression with no reversals: 8/16/20 terms maps onto retention 0.93443/0.93489/0.93514 and char 80.896/80.656/80.274",
"coverage_predicted_ordering": "coverage got the ORDERING right here even though it has never predicted magnitudes"
},
"combined_multiseed": {
"results": {
"per_seed": [
{
"seed": 11,
"shipped": {
"char_pct": 79.119,
"term_10dp": 0.931462117
},
"combined": {
"char_pct": 81.129,
"term_10dp": 0.9309158796
}
},
{
"seed": 29,
"shipped": {
"char_pct": 76.17,
"term_10dp": 0.945231517
},
"combined": {
"char_pct": 78.369,
"term_10dp": 0.9448618065
}
},
{
"seed": 47,
"shipped": {
"char_pct": 78.453,
"term_10dp": 0.933831047
},
"combined": {
"char_pct": 80.695,
"term_10dp": 0.9331452292
}
},
{
"seed": 73,
"shipped": {
"char_pct": 77.266,
"term_10dp": 0.9440895859
},
"combined": {
"char_pct": 79.43,
"term_10dp": 0.9431722632
}
},
{
"seed": 101,
"shipped": {
"char_pct": 79.151,
"term_10dp": 0.9301182346
},
"combined": {
"char_pct": 81.241,
"term_10dp": 0.9298644773
}
}
],
"term_delta_mean": -0.0005545691,
"term_delta_stdev": 0.0002615082,
"char_delta_mean": 2.141,
"char_delta_stdev": 0.092,
"seeds_losing_retention": 5
},
"significance": "2.12 sigma, losing on all 5 seeds"
},
"gate": {
"disabled_identical_to_shipped": 150,
"invariants": {
"unbalanced": 0,
"tildes_even": true,
"grew": 0,
"backrefs": 315,
"unresolved": 0
},
"decoder_forward": {
"role_recovery": 1.0,
"tool_recovery": 1.0,
"pair_recovery": 1.0,
"order_fidelity": 1.0,
"challenges": 150,
"event_count_matches": 150
},
"decoder_combined": {
"role_recovery": 1.0,
"tool_recovery": 1.0,
"pair_recovery": 1.0,
"order_fidelity": 1.0,
"challenges": 150,
"event_count_matches": 150
}
},
"decision": "shipped remains the lead; forward and combined are NOT promoted and default OFF. The combined rule is the best aggressive policy on record (+2.141 char for -0.00055 term). Still a trade, and I cannot convert a retention delta into run-failure probability - the 7.6% break-even from v82 is unmeasurable without an OpenRouter key",
"what_changed": "the frontier is now MAPPED - four points, monotone, multi-seed variance on each - so a future decision with a real retention-to-failure mapping can pick a point off this curve instead of re-deriving it",
"revert_proofs": "22 checks: 3 LIVE source mutations (combined default enabled, forward default enabled, union collapsed to the subsumed doc-unique conjunction) and 19 report mutations, all firing"
}