Report-Genius / backend /observability /token_audit.py
StormShadow308's picture
Deploy RICS v2 backend (CPU embedder + reranker, baked jina models)
a671976
Raw
History Blame Contribute Delete
2.87 kB
"""Tiktoken-based token counting for embedder / reranker audit trails.
Uses ``cl100k_base`` (same encoding as the app stack prompt builder). jina-
embeddings-v3 and jina-reranker-v3 use an XLM-RoBERTa tokenizer internally, so
these counts are an **audit proxy** — good for comparing relative payload size,
spotting truncation risk, and tracking regressions. They will not match the
model tokenizer exactly.
Best-effort only: if tiktoken is missing, counts fall back to a whitespace
word estimate so retrieval never breaks.
"""
from __future__ import annotations
import logging
from functools import lru_cache
from typing import TypedDict
logger = logging.getLogger(__name__)
ENCODING_NAME = "cl100k_base"
class TextTokenSummary(TypedDict):
chars: int
tokens: int
class EmbedderFeedSummary(TextTokenSummary):
max_seq_length: int
would_truncate: bool
class RerankerFeedSummary(TypedDict):
full_chars: int
full_tokens: int
fed_chars: int
fed_tokens: int
doc_chars_cap: int
@lru_cache(maxsize=1)
def _encoding():
import tiktoken
return tiktoken.get_encoding(ENCODING_NAME)
def count_tokens(text: str) -> int:
"""Return tiktoken token count for ``text`` (0 for empty)."""
raw = text or ""
if not raw.strip():
return 0
try:
return len(_encoding().encode(raw))
except Exception: # noqa: BLE001 - audit helper must not break callers
logger.debug("tiktoken encode failed; using word fallback", exc_info=True)
return max(1, len(raw.split()))
def summarize_text(text: str) -> TextTokenSummary:
raw = text or ""
return {"chars": len(raw), "tokens": count_tokens(raw)}
def summarize_embedder_feed(text: str, *, max_seq_length: int) -> EmbedderFeedSummary:
"""Summarize text as fed to the local embedder (truncated at ``max_seq_length``)."""
base = summarize_text(text)
cap = max(0, int(max_seq_length or 0))
would_truncate = bool(cap and base["tokens"] > cap)
return {
**base,
"max_seq_length": cap,
"would_truncate": would_truncate,
}
def reranker_fed_text(text: str, *, doc_chars_cap: int) -> str:
"""Text slice actually passed to jina-reranker-v3 (char cap, not token cap)."""
cap = max(0, int(doc_chars_cap or 0))
raw = text or ""
return raw if not cap else raw[:cap]
def summarize_reranker_feed(text: str, *, doc_chars_cap: int) -> RerankerFeedSummary:
"""Full chunk vs char-capped payload fed to the reranker."""
fed = reranker_fed_text(text, doc_chars_cap=doc_chars_cap)
full = summarize_text(text)
fed_summary = summarize_text(fed)
return {
"full_chars": full["chars"],
"full_tokens": full["tokens"],
"fed_chars": fed_summary["chars"],
"fed_tokens": fed_summary["tokens"],
"doc_chars_cap": int(doc_chars_cap or 0),
}