multi-agent-system / tests /test_rag_quality.py
firepenguindisopanda
Add comprehensive documentation and guidelines for various roles
b7e4cb3
Raw
History Blame Contribute Delete
12.4 kB
"""RAG quality evaluation: content coverage, keyword hit rate, retrieval quality."""
from pathlib import Path
from collections import Counter
from app.core.rag import CORPUS_DIR
from app.core.schemas import TeamRole
# ── Corpus file structure check ───────────────────────────────────────
ROLES_WITH_RAG = [
TeamRole.PRODUCT_OWNER,
TeamRole.BUSINESS_ANALYST,
TeamRole.SOLUTION_ARCHITECT,
TeamRole.DATA_ARCHITECT,
TeamRole.SECURITY_ANALYST,
TeamRole.UX_DESIGNER,
TeamRole.API_DESIGNER,
TeamRole.QA_STRATEGIST,
TeamRole.DEVOPS_ARCHITECT,
TeamRole.TECHNICAL_WRITER,
]
ROLE_DIR_MAP: dict[TeamRole, str] = {
TeamRole.PRODUCT_OWNER: "product_owner",
TeamRole.BUSINESS_ANALYST: "business_analyst",
TeamRole.SOLUTION_ARCHITECT: "solution_architect",
TeamRole.DATA_ARCHITECT: "data_architect",
TeamRole.SECURITY_ANALYST: "security_analyst",
TeamRole.UX_DESIGNER: "ux_designer",
TeamRole.API_DESIGNER: "api_designer",
TeamRole.QA_STRATEGIST: "qa_strategist",
TeamRole.DEVOPS_ARCHITECT: "devops_architect",
TeamRole.TECHNICAL_WRITER: "technical_writer",
}
MIN_FILES_PER_ROLE = 3
MIN_LINES_PER_ROLE = 60
# ── Keyword coverage per role ─────────────────────────────────────────
# Topics that each role's corpus should cover, based on the role's domain
ROLE_TOPICS: dict[TeamRole, list[str]] = {
TeamRole.PRODUCT_OWNER: [
"backlog", "prioritiz", "stakeholder", "value", "moscow",
"mvp", "scope", "kano",
],
TeamRole.BUSINESS_ANALYST: [
"requirement", "elicitation", "use case", "interview", "workshop",
"stakeholder", "user story",
],
TeamRole.SOLUTION_ARCHITECT: [
"architecture", "trade-off", "scal", "pattern", "case study",
"decomposition", "latency", "caching",
],
TeamRole.DATA_ARCHITECT: [
"data model", "schema", "storage", "sql", "nosql",
"normalization", "partition", "index",
],
TeamRole.SECURITY_ANALYST: [
"threat model", "compliance", "authentication", "oauth",
"encryption", "owasp", "vulnerability",
],
TeamRole.UX_DESIGNER: [
"navigation", "accessibility", "wcag", "feedback",
"interaction", "touch target", "empty state",
],
TeamRole.API_DESIGNER: [
"versioning", "error handling", "rest", "status code",
"idempotent", "rate limit", "pagination",
],
TeamRole.QA_STRATEGIST: [
"test strategy", "unit test", "integration", "e2e",
"quality metric", "coverage", "regression", "flaky",
],
TeamRole.DEVOPS_ARCHITECT: [
"ci/cd", "deployment", "docker", "kubernetes",
"monitoring", "rollback", "infrastructure",
],
TeamRole.TECHNICAL_WRITER: [
"documentation", "api doc", "style guide", "voice",
"user guide", "readme", "changelog",
],
}
def _load_role_texts(role: TeamRole) -> list[str]:
"""Load all text from a role's corpus directory."""
dir_name = ROLE_DIR_MAP.get(role, role.value)
role_dir = CORPUS_DIR / dir_name
if not role_dir.is_dir():
return []
texts: list[str] = []
for path in sorted(role_dir.iterdir()):
if path.suffix in {".txt", ".md", ".yaml", ".yml"}:
texts.append(path.read_text(encoding="utf-8"))
return texts
def _keyword_overlap(query: str, document: str) -> float:
"""Measure what fraction of query words appear in the document."""
query_words = set(query.lower().split())
doc_words = set(document.lower().split())
if not query_words:
return 0.0
return len(query_words & doc_words) / len(query_words)
def _hit_rate(results: list[bool]) -> float:
return sum(results) / len(results) if results else 0.0
# ── Tests ─────────────────────────────────────────────────────────────
class TestCorpusStructure:
"""Validate basic corpus structure for all roles."""
def test_corpus_dir_exists(self) -> None:
assert CORPUS_DIR.is_dir(), f"Corpus dir not found: {CORPUS_DIR}"
def test_all_roles_have_directories(self) -> None:
missing: list[str] = []
for role, dir_name in ROLE_DIR_MAP.items():
role_dir = CORPUS_DIR / dir_name
if not role_dir.is_dir():
missing.append(f"{role.value} ({dir_name})")
assert not missing, f"Missing corpus directories: {missing}"
def test_no_orphan_directories(self) -> None:
expected_dirs = {d for d in ROLE_DIR_MAP.values()}
actual_dirs = {d.name for d in CORPUS_DIR.iterdir() if d.is_dir()}
orphans = actual_dirs - expected_dirs
assert not orphans, f"Orphaned corpus directories: {orphans}"
def test_each_role_meets_min_file_count(self) -> None:
failures: list[str] = []
for role in ROLES_WITH_RAG:
texts = _load_role_texts(role)
if len(texts) < MIN_FILES_PER_ROLE:
failures.append(f"{role.value}: {len(texts)} files (min {MIN_FILES_PER_ROLE})")
assert not failures, f"Role file count below minimum:\n" + "\n".join(failures)
def test_each_role_meets_min_line_count(self) -> None:
failures: list[str] = []
for role in ROLES_WITH_RAG:
texts = _load_role_texts(role)
total_lines = sum(t.count("\n") + 1 for t in texts)
if total_lines < MIN_LINES_PER_ROLE:
failures.append(f"{role.value}: {total_lines} lines (min {MIN_LINES_PER_ROLE})")
assert not failures, f"Role line count below minimum:\n" + "\n".join(failures)
class TestTopicCoverage:
"""Validate that each role's corpus covers expected topics."""
def test_all_roles_cover_expected_topics(self) -> None:
failures: list[str] = []
for role in ROLES_WITH_RAG:
texts = _load_role_texts(role)
combined = "\n".join(texts).lower()
missing_topics: list[str] = []
for topic in ROLE_TOPICS.get(role, []):
if topic.lower() not in combined:
missing_topics.append(topic)
if missing_topics:
failures.append(f"{role.value}: missing topics: {missing_topics}")
assert not failures, "\n" + "\n".join(failures)
def test_topic_coverage_percentage(self) -> None:
"""Measure and report topic coverage percentage per role."""
report: list[str] = []
for role in ROLES_WITH_RAG:
texts = _load_role_texts(role)
combined = "\n".join(texts).lower()
topics = ROLE_TOPICS.get(role, [])
covered = sum(1 for t in topics if t.lower() in combined)
pct = (covered / len(topics) * 100) if topics else 0
report.append(f" {role.value}: {covered}/{len(topics)} ({pct:.0f}%)")
print("Topic coverage by role:\n" + "\n".join(report))
class TestKeywordHitRate:
"""Simple keyword-based retrieval quality eval (proxy for semantic similarity)."""
QUERIES: dict[TeamRole, list[tuple[str, str]]] = {
TeamRole.PRODUCT_OWNER: [
("how to prioritize backlog items using moscow method", "backlog_prioritization"),
("how to make value-driven product decisions", "value_driven_decisions"),
],
TeamRole.BUSINESS_ANALYST: [
("techniques for gathering requirements from stakeholders", "requirements_elicitation"),
("how to write well-structured use cases", "use_case_writing"),
],
TeamRole.SOLUTION_ARCHITECT: [
("how to evaluate architectural tradeoffs and patterns", "patterns_and_tradeoffs"),
("case study for scaling an application", "case_study_scaling"),
],
TeamRole.DATA_ARCHITECT: [
("patterns for data modeling and schema design", "data_modeling_patterns"),
("deciding between sql and nosql storage", "storage_decisions"),
],
TeamRole.SECURITY_ANALYST: [
("threat modeling approaches and techniques", "threat_modeling"),
("compliance frameworks for security", "compliance_frameworks"),
],
TeamRole.API_DESIGNER: [
("strategies for versioning rest apis", "api_versioning_strategies"),
("error handling patterns for api responses", "error_handling_patterns"),
],
TeamRole.UX_DESIGNER: [
("ui interaction patterns for web applications", "interaction_patterns"),
("wcag accessibility standards and best practices", "accessibility_standards"),
],
TeamRole.QA_STRATEGIST: [
("test strategy patterns and the testing pyramid", "test_strategy_patterns"),
("quality metrics for measuring test effectiveness", "quality_metrics"),
],
TeamRole.DEVOPS_ARCHITECT: [
("ci cd best practices for continuous deployment", "cicd_best_practices"),
("deployment patterns for production systems", "deployment_patterns"),
],
TeamRole.TECHNICAL_WRITER: [
("types of documentation and when to use them", "documentation_types"),
("style and tone guidelines for technical writing", "style_and_tone"),
],
}
def test_keyword_hit_rate_meets_threshold(self) -> None:
"""For each query, check that the most relevant corpus doc
has non-trivial keyword overlap with the query."""
failures: list[str] = []
all_queries = 0
hits = 0
for role in ROLES_WITH_RAG:
texts = _load_role_texts(role)
if not texts:
continue
queries = self.QUERIES.get(role, [])
for query, expected_substring in queries:
all_queries += 1
scores = [_keyword_overlap(query, t) for t in texts]
best_score = max(scores) if scores else 0.0
# Find which doc best completed the query - check if expected file contributed
expected_hit = any(expected_substring in t[:200].lower()
for t in texts)
if best_score >= 0.1 or expected_hit:
hits += 1
else:
failures.append(
f"{role.value}: query '{query[:50]}...' "
f"best overlap {best_score:.2f}"
)
hit_pct = _hit_rate([True] * hits + [False] * (all_queries - hits)) * 100
print(f"Keyword hit rate: {hits}/{all_queries} ({hit_pct:.0f}%)")
if hit_pct < 70:
failures.insert(0, f"Overall hit rate {hit_pct:.0f}% < 70% threshold")
assert not failures, "\n" + "\n".join(failures)
class TestRoleScopedRetrieval:
"""Verify that role-scoped retrieval returns more relevant docs than global."""
QUERIES: dict[TeamRole, str] = {
TeamRole.PRODUCT_OWNER: "how to prioritize features in the backlog",
TeamRole.SECURITY_ANALYST: "threat modeling and security compliance",
TeamRole.DEVOPS_ARCHITECT: "ci cd deployment pipeline best practices",
}
def test_role_scoped_outperforms_global(self) -> None:
"""Role-scoped context should have higher keyword overlap than global."""
failures: list[str] = []
for role, query in self.QUERIES.items():
role_texts = _load_role_texts(role)
if not role_texts:
continue
role_combined = "\n".join(role_texts)
role_score = _keyword_overlap(query, role_combined)
# Global: combine all roles' text and measure
all_texts: list[str] = []
for r in ROLES_WITH_RAG:
if r != role:
all_texts.extend(_load_role_texts(r))
global_combined = "\n".join(all_texts)
global_score = _keyword_overlap(query, global_combined)
if role_score <= global_score:
failures.append(
f"{role.value}: role score {role_score:.2f} <= global {global_score:.2f}"
)
assert not failures, "\n".join(failures)