Spaces:
Sleeping
Sleeping
| """Finite, metric-safe vocabulary for structured-RAG observability. | |
| Local operation records retain every safe caller-supplied key for diagnosis. | |
| Only the names in this module may become span attributes or metric dimensions, | |
| which prevents instrumentation drift from creating unbounded SigNoz series. | |
| """ | |
| from __future__ import annotations | |
| RAG_STAGES = frozenset({ | |
| "pdf_parse", "canonical_store", "dense_index_write", "project_memory_write", | |
| "anchor_resolution", "lexical_search", | |
| "dense_search", "rank_fusion", "structural_expansion", "reranking", | |
| "diversity_selection", "context_packing", "citation_materialization", | |
| "project_memory_recall", "student_memory_recall", "source_context", | |
| "context_composition", "memory_promotion", "memory_flush", "rebuild_document", | |
| }) | |
| EVIDENCE_TYPES = frozenset({ | |
| "title", "heading", "paragraph", "list", "table", "figure", "plot", | |
| "diagram", "formula", "caption", "footnote", "reference", "annotation", | |
| "section_card", "local_window", | |
| }) | |
| # Concrete bounded measurements emitted by the structured RAG, memory, and | |
| # Cerebras operation boundaries. The two evidence-type families are normalized | |
| # separately below so their suffixes remain limited to ``EVIDENCE_TYPES``. | |
| RAG_MEASURES = frozenset({ | |
| "anchor_count", | |
| "context_token_budget", | |
| "context_truncated", | |
| "datasets_ok", | |
| "datasets_total", | |
| "dense_candidate_count", | |
| "documents_represented", | |
| "documents_failed", | |
| "documents_succeeded", | |
| "duplicates_removed", | |
| "embedding_batch_size", | |
| "empty_result", | |
| "evidence_unit_count", | |
| "fused_candidate_count", | |
| "indexed_unit_count", | |
| "lexical_candidate_count", | |
| "llm_attempts", | |
| "llm_chunks_approx", | |
| "llm_tokens", | |
| "page_count", | |
| "project_candidate_count", | |
| "project_candidates_with_source_evidence", | |
| "project_memory_count", | |
| "project_memory_chars", | |
| "quality_flag_count", | |
| "query_chars", | |
| "returned_evidence_count", | |
| "sections_represented", | |
| "source_context_chars", | |
| "source_evidence_count", | |
| "student_candidate_count", | |
| "student_candidate_promoted_count", | |
| "student_candidate_rejected_count", | |
| "student_memory_chars", | |
| "student_memory_count", | |
| }) | |
| def normalize_stage(name: str) -> str | None: | |
| """Return a registered stage name, or ``None`` for vocabulary drift.""" | |
| value = str(name).strip().lower() | |
| return value if value in RAG_STAGES else None | |
| def normalize_measure(name: str) -> str | None: | |
| """Return a registered measurement name, or ``None`` when unbounded.""" | |
| value = str(name).strip().lower() | |
| if value.startswith(("evidence_type.", "returned_type.")): | |
| prefix, suffix = value.split(".", 1) | |
| return f"{prefix}.{suffix}" if suffix in EVIDENCE_TYPES else None | |
| return value if value in RAG_MEASURES else None | |