| |
| """Resolve =rN dedup back-references and verify each points at a real payload. |
| |
| This was the last decode path never tested. decode_eval checks role/tool/pair/ |
| order and event counts; thinking_decode_eval checks thinking spans; neither |
| resolves a back-reference. _squeeze_tool_result_dedup emits "=rN" pointing at an |
| earlier tool_result, and nothing had confirmed a decoder can follow it. |
| |
| RESULT: the encoder is correct, 315 references and 0 unresolvable. _id() assigns |
| ONE label shared by a tool_call and its matching tool_result, so the |
| tool_result ordinal used by the dedup counter and the sigil label coincide by |
| construction. |
| |
| PARSING WARNING, learned the hard way: a sigil body is NOT delimited by the next |
| guillemet in a naive way, and the separator after the label is not always a |
| space (it can be a newline). Three of my own parsers reported false failure rates |
| of 96.2%, 77.8% and 74.9% before this one read the format correctly. |
| """ |
| from __future__ import annotations |
| import argparse, csv, importlib.util, json, re |
| from pathlib import Path |
|
|
| REPO = Path("/var/lib/octave/sn114/repo"); PLUG = Path("/var/lib/octave/sn114/external/SOMA-plugin") |
| csv.field_size_limit(50_000_000) |
| OPEN = chr(0x00ab); CLOSE = chr(0x00bb) |
| SIGIL = re.compile(OPEN + r"r(\d+)[ ]?([^" + OPEN + CLOSE + r"]*)") |
|
|
|
|
| def parse_results(out: str) -> dict[str, str]: |
| """label -> emitted body. Optional separator, body ends at the next sigil.""" |
| res: dict[str, str] = {} |
| for m in SIGIL.finditer(out): |
| res.setdefault(m.group(1), m.group(2)) |
| return res |
|
|
|
|
| def main() -> None: |
| ap = argparse.ArgumentParser() |
| ap.add_argument("--compressor", default="structural_cot_compressor.py") |
| ap.add_argument("--limit", type=int, default=150) |
| ap.add_argument("--out", type=Path, required=True) |
| a = ap.parse_args() |
| dd = REPO/"miner/plain_text_compression/sample_tasks/CoT-Compression-1" |
| rows = list(csv.DictReader((dd/"challenges.csv").open()))[:a.limit] |
| s = importlib.util.spec_from_file_location("c", PLUG/a.compressor) |
| m = importlib.util.module_from_spec(s); s.loader.exec_module(m) |
|
|
| total = unresolvable = chained = 0 |
| for r in rows: |
| by = parse_results(m.compress_content(r["challenge_text"])) |
| for lbl, body in by.items(): |
| mr = re.fullmatch(r"=r(\d+)", body.strip()) |
| if not mr: |
| continue |
| total += 1 |
| tgt = mr.group(1) |
| if tgt not in by: |
| unresolvable += 1 |
| elif by[tgt].strip().startswith("=r"): |
| chained += 1 |
| res = {"compressor": a.compressor, "challenges": len(rows), |
| "references": total, "unresolvable": unresolvable, |
| "chained_references": chained, |
| "resolution_rate": round(1 - unresolvable/total, 4) if total else 1.0} |
| a.out.write_text(json.dumps(res, indent=2)) |
| print(a.compressor, res["references"], "refs,", unresolvable, "unresolvable,", |
| chained, "chained") |
|
|
|
|
| if __name__ == "__main__": |
| main() |