| """Frozen E02 context packers over immutable E01 rankings.""" |
|
|
| from __future__ import annotations |
|
|
| import json |
| from pathlib import Path |
| import re |
| from typing import Any, Sequence |
|
|
| from .components import Candidate |
| from .repository import GitSnapshot |
| from .syntax_index import parse_go_file |
| from .tokenization import QwenTokenCounter |
|
|
|
|
| EVIDENCE_TOKEN_BUDGET = 60_000 |
|
|
|
|
| class PackingError(RuntimeError): |
| """Raised when frozen retrieval evidence cannot be packed safely.""" |
|
|
|
|
| def ranking_path(root: Path, experiment: str, harness: str, task: str) -> Path: |
| matches = sorted((root / "results" / "raw" / experiment / harness / task).glob("*/ranking.json")) |
| if len(matches) != 1: |
| raise PackingError( |
| f"expected one immutable {experiment}/{harness}/{task} ranking, found {len(matches)}" |
| ) |
| return matches[0] |
|
|
|
|
| def load_ranking_records(path: Path, limit: int = 200) -> list[dict[str, Any]]: |
| value = json.loads(path.read_text(encoding="utf-8")) |
| if not isinstance(value, list): |
| raise PackingError(f"ranking is not an array: {path}") |
| records: list[dict[str, Any]] = [] |
| for item in value[:limit]: |
| if not isinstance(item, dict): |
| raise PackingError(f"ranking item is not an object: {path}") |
| |
| records.append( |
| { |
| "rank": int(item["rank"]), |
| "path": str(item["path"]), |
| "line_start": int(item["line_start"]), |
| "line_end": int(item["line_end"]), |
| "score": float(item["score"]), |
| "source": str(item["source"]), |
| "symbol": item.get("symbol"), |
| } |
| ) |
| return records |
|
|
|
|
| def candidates_from_records( |
| snapshot: GitSnapshot, |
| commit: str, |
| records: Sequence[dict[str, Any]], |
| ) -> tuple[Candidate, ...]: |
| candidates: list[Candidate] = [] |
| for item in records: |
| source = snapshot.read_file(commit, item["path"]) |
| lines = source.text.splitlines() |
| start = max(item["line_start"], 1) |
| end = min(item["line_end"], len(lines)) |
| text = "\n".join(lines[start - 1 : end]) |
| candidates.append( |
| Candidate( |
| path=item["path"], |
| line_start=start, |
| line_end=end, |
| text=text, |
| source=item["source"], |
| score=item["score"], |
| symbol=item.get("symbol"), |
| ) |
| ) |
| return tuple(candidates) |
|
|
|
|
| def pack_snippets( |
| tokenizer: QwenTokenCounter, |
| candidates: Sequence[Candidate], |
| budget: int = EVIDENCE_TOKEN_BUDGET, |
| ) -> tuple[str, tuple[str, ...], int]: |
| text, included, used = tokenizer.pack_ranked(candidates, budget) |
| return text, tuple(dict.fromkeys(item.path for item in included)), used |
|
|
|
|
| def _pack_blocks( |
| tokenizer: QwenTokenCounter, |
| blocks: Sequence[tuple[str, str]], |
| budget: int, |
| ) -> tuple[str, tuple[str, ...], int]: |
| selected: list[str] = [] |
| paths: list[str] = [] |
| used = 0 |
| for path, block in blocks: |
| count = tokenizer.count(block) |
| if used + count > budget: |
| continue |
| selected.append(block) |
| paths.append(path) |
| used += count |
| return "\n".join(selected), tuple(dict.fromkeys(paths)), used |
|
|
|
|
| def pack_skeletons( |
| tokenizer: QwenTokenCounter, |
| snapshot: GitSnapshot, |
| commit: str, |
| paths: Sequence[str], |
| budget: int = EVIDENCE_TOKEN_BUDGET, |
| ) -> tuple[str, tuple[str, ...], int]: |
| blocks: list[tuple[str, str]] = [] |
| for path in dict.fromkeys(paths): |
| source = snapshot.read_file(commit, path) |
| symbols = parse_go_file(path, source.text) |
| signatures = "\n".join( |
| f"{item.kind} {item.name} lines {item.line_start}-{item.line_end}: {item.signature}" |
| for item in symbols |
| ) |
| blocks.append((path, f"\n--- SYMBOL SKELETON: {path} ---\n{signatures}\n")) |
| return _pack_blocks(tokenizer, blocks, budget) |
|
|
|
|
| def pack_whole_files( |
| tokenizer: QwenTokenCounter, |
| snapshot: GitSnapshot, |
| commit: str, |
| paths: Sequence[str], |
| budget: int = EVIDENCE_TOKEN_BUDGET, |
| ) -> tuple[str, tuple[str, ...], int]: |
| blocks = [ |
| (path, f"\n--- WHOLE FILE: {path} ---\n{snapshot.read_file(commit, path).text}\n") |
| for path in dict.fromkeys(paths) |
| ] |
| return _pack_blocks(tokenizer, blocks, budget) |
|
|
|
|
| def pack_role_summaries( |
| tokenizer: QwenTokenCounter, |
| snapshot: GitSnapshot, |
| commit: str, |
| paths: Sequence[str], |
| budget: int = EVIDENCE_TOKEN_BUDGET, |
| ) -> tuple[str, tuple[str, ...], int]: |
| blocks: list[tuple[str, str]] = [] |
| for path in dict.fromkeys(paths): |
| source = snapshot.read_file(commit, path) |
| package = re.search(r"(?m)^package\s+(\w+)", source.text) |
| symbols = parse_go_file(path, source.text) |
| kinds: dict[str, list[str]] = {} |
| for symbol in symbols: |
| kinds.setdefault(symbol.kind, []).append(symbol.name) |
| summary = [f"package: {package.group(1) if package else 'unknown'}"] |
| summary.extend(f"{kind}: {', '.join(names)}" for kind, names in sorted(kinds.items())) |
| blocks.append((path, f"\n--- ROLE SUMMARY: {path} ---\n" + "\n".join(summary) + "\n")) |
| return _pack_blocks(tokenizer, blocks, budget) |
|
|
|
|
| def pack_specialized_channels( |
| root: Path, |
| task_id: str, |
| snapshot: GitSnapshot, |
| commit: str, |
| tokenizer: QwenTokenCounter, |
| budget: int = EVIDENCE_TOKEN_BUDGET, |
| ) -> tuple[str, tuple[str, ...], int]: |
| channels = ( |
| ("LEXICAL SEARCH", "H001"), |
| ("TREE-SITTER SYMBOL SEARCH", "H002"), |
| ("SEMANTIC SEARCH", "H003"), |
| ) |
| per_channel = budget // len(channels) |
| texts: list[str] = [] |
| paths: list[str] = [] |
| used = 0 |
| for label, harness in channels: |
| records = load_ranking_records(ranking_path(root, "E01", harness, task_id)) |
| candidates = candidates_from_records(snapshot, commit, records) |
| text, included, tokens = pack_snippets(tokenizer, candidates, per_channel) |
| texts.append(f"\n====== {label} RESULTS ======\n{text}") |
| paths.extend(included) |
| used += tokens |
| return "".join(texts), tuple(dict.fromkeys(paths)), used |
|
|