Download src/peppa/evidence.py from ChatterjeeLab/PepPA: direct link, hf CLI and curl.
- Browser
- Download file 1.18 kB
-
https://huggingface.co/ChatterjeeLab/PepPA/resolve/main/src/peppa/evidence.py
- Command line
-
hf download hf://ChatterjeeLab/PepPA/src/peppa/evidence.py
-
curl -L -o evidence.py https://huggingface.co/ChatterjeeLab/PepPA/resolve/main/src/peppa/evidence.py
1.18 kB
| """Local passage retrieval and source-grounded claim validation.""" | |
| import json | |
| import re | |
| from pathlib import Path | |
| from .schema import Evidence, ToolResult | |
| def search_local(path, query, limit=8): | |
| """Rank source passages by query token overlap and retain complete records.""" | |
| terms=set(re.findall(r"[a-z0-9]+",query.lower())) | |
| records=[Evidence.model_validate(json.loads(x)) for x in Path(path).read_text().splitlines() if x.strip()] | |
| scored=[(len(terms & set(re.findall(r"[a-z0-9]+",r.passage.lower()+" "+r.claim.lower()))),r) for r in records] | |
| return [r for s,r in sorted(scored,key=lambda x:(-x[0],x[1].id))[:limit] if s>0] | |
| def retrieval_tool(path): | |
| def run(arguments,state): | |
| records=search_local(path,arguments["query"],int(arguments.get("limit",8))) | |
| return ToolResult(evidence=records,message=f"Retrieved {len(records)} source records") | |
| return run | |
| def validate_claim(evidence: Evidence, source_text: str): | |
| """Require the stored passage to occur verbatim in its retained source.""" | |
| if evidence.passage not in source_text: | |
| raise ValueError("evidence passage is absent from the source snapshot") | |
| return evidence | |