Hybrid_Rag_System / response_evaluation_agent.py
MaxyZand's picture
Upload 15 files
f565efa verified
Raw
History Blame Contribute Delete
4.38 kB
"""
Response Evaluation Agent.
Scores the generated answer against five metrics and decides:
Metric Threshold
Relevance > 0.85
Groundedness > 0.90
Hallucination Rate 0 (must be exactly 0)
Completeness > 0.85
Citation Present
- All thresholds met -> "pass"
- Relevance fails -> "re_retrieve" (the evidence itself wasn't
on-topic — that's a retrieval problem, no
amount of rewriting the answer fixes it)
- Relevance OK, but one of the other four fails
-> "rewrite" (the evidence was fine; the
generated answer didn't use it well — a
generation problem)
Relevance and Groundedness are heuristics built from data the rest of
the pipeline already computed (rerank scores, hallucination rate) rather
than separate model calls. Completeness is judged by the LLM, since
"did this fully answer the question" isn't something a similarity score
captures well.
"""
import re
from typing import Any, Dict, List
import numpy as np
from langchain_core.documents import Document
THRESHOLDS = {
"relevance": 0.85,
"groundedness": 0.90,
"hallucination_rate": 0.0,
"completeness": 0.85,
}
COMPLETENESS_PROMPT = """Rate how completely the ANSWER addresses every part of the QUESTION.
Return ONLY a single number between 0 and 1 (no words, no explanation).
1.0 means every part of the question is fully addressed.
0.0 means the question is not addressed at all.
QUESTION:
{query}
ANSWER:
{answer}
Score:
"""
def _sigmoid(x: float) -> float:
return 1.0 / (1.0 + np.exp(-x))
def _relevance_score(evidence_docs: List[Document]) -> float:
"""
Heuristic: average the reranker's cross-encoder scores (already on
each doc's metadata as rerank_score) and squash to (0, 1) with a
sigmoid, since raw cross-encoder scores aren't bounded.
"""
scores = [doc.metadata.get("rerank_score") for doc in evidence_docs if doc.metadata.get("rerank_score") is not None]
if not scores:
return 0.0
return round(_sigmoid(sum(scores) / len(scores)), 3)
def _completeness_score(llm, query: str, answer: str) -> float:
prompt = COMPLETENESS_PROMPT.format(query=query, answer=answer)
try:
response = llm.invoke(prompt)
match = re.search(r"(0(\.\d+)?|1(\.0+)?)", response.content)
return round(float(match.group()), 3) if match else 0.0
except (AttributeError, ValueError):
return 0.0
def evaluate_response(
llm,
query: str,
answer: str,
evidence_docs: List[Document],
hallucination_report: Dict[str, Any],
citation_result: Dict[str, Any],
) -> Dict[str, Any]:
"""
Returns:
{
"metrics": {"relevance": .., "groundedness": .., "hallucination_rate": ..,
"completeness": .., "citation_present": bool},
"checks": {"relevance": bool, "groundedness": bool, "hallucination_rate": bool,
"completeness": bool, "citation": bool},
"passed": bool,
"decision": "pass" | "rewrite" | "re_retrieve",
}
"""
relevance = _relevance_score(evidence_docs)
hallucination_rate = hallucination_report["hallucination_rate"]
groundedness = round(1 - hallucination_rate, 3)
completeness = _completeness_score(llm, query, answer)
citation_present = citation_result["citation_present"]
checks = {
"relevance": relevance > THRESHOLDS["relevance"],
"groundedness": groundedness > THRESHOLDS["groundedness"],
"hallucination_rate": hallucination_rate <= THRESHOLDS["hallucination_rate"],
"completeness": completeness > THRESHOLDS["completeness"],
"citation": citation_present,
}
passed = all(checks.values())
if passed:
decision = "pass"
elif not checks["relevance"]:
decision = "re_retrieve"
else:
decision = "rewrite"
return {
"metrics": {
"relevance": relevance,
"groundedness": groundedness,
"hallucination_rate": hallucination_rate,
"completeness": completeness,
"citation_present": citation_present,
},
"checks": checks,
"passed": passed,
"decision": decision,
}