"""Finite, metric-safe vocabulary for structured-RAG observability. Local operation records retain every safe caller-supplied key for diagnosis. Only the names in this module may become span attributes or metric dimensions, which prevents instrumentation drift from creating unbounded SigNoz series. """ from __future__ import annotations RAG_STAGES = frozenset({ "pdf_parse", "canonical_store", "dense_index_write", "project_memory_write", "anchor_resolution", "lexical_search", "dense_search", "rank_fusion", "structural_expansion", "reranking", "diversity_selection", "context_packing", "citation_materialization", "project_memory_recall", "student_memory_recall", "source_context", "context_composition", "memory_promotion", "memory_flush", "rebuild_document", }) EVIDENCE_TYPES = frozenset({ "title", "heading", "paragraph", "list", "table", "figure", "plot", "diagram", "formula", "caption", "footnote", "reference", "annotation", "section_card", "local_window", }) # Concrete bounded measurements emitted by the structured RAG, memory, and # Cerebras operation boundaries. The two evidence-type families are normalized # separately below so their suffixes remain limited to ``EVIDENCE_TYPES``. RAG_MEASURES = frozenset({ "anchor_count", "context_token_budget", "context_truncated", "datasets_ok", "datasets_total", "dense_candidate_count", "documents_represented", "documents_failed", "documents_succeeded", "duplicates_removed", "embedding_batch_size", "empty_result", "evidence_unit_count", "fused_candidate_count", "indexed_unit_count", "lexical_candidate_count", "llm_attempts", "llm_chunks_approx", "llm_tokens", "page_count", "project_candidate_count", "project_candidates_with_source_evidence", "project_memory_count", "project_memory_chars", "quality_flag_count", "query_chars", "returned_evidence_count", "sections_represented", "source_context_chars", "source_evidence_count", "student_candidate_count", "student_candidate_promoted_count", "student_candidate_rejected_count", "student_memory_chars", "student_memory_count", }) def normalize_stage(name: str) -> str | None: """Return a registered stage name, or ``None`` for vocabulary drift.""" value = str(name).strip().lower() return value if value in RAG_STAGES else None def normalize_measure(name: str) -> str | None: """Return a registered measurement name, or ``None`` when unbounded.""" value = str(name).strip().lower() if value.startswith(("evidence_type.", "returned_type.")): prefix, suffix = value.split(".", 1) return f"{prefix}.{suffix}" if suffix in EVIDENCE_TYPES else None return value if value in RAG_MEASURES else None