AudioDecisionModel / server_runtime /transcript_keyword.py
gojiteji's picture
Audio Decision Model demo
148af80
Raw
History Blame Contribute Delete
11 kB
"""Validation-frozen free-form keyword decisions over one shared ASR transcript."""
from __future__ import annotations
from dataclasses import dataclass
from functools import lru_cache
import math
import re
from typing import Any, Mapping
import unicodedata
from sudachipy import dictionary, tokenizer
from .typed_audio import question_options, typed_answer
MATCH_THRESHOLD = 0.60
MINIMUM_FUZZY_LENGTH = 2
TASKS = {"keyword_presence", "keyword_choice", "keyword_count"}
_CUES = (
"含まれ", "語句", "キーワード", "発話された語", "言っていますか", "言った",
"contain", "keyword", "phrase", "spoken word", "say ", "said ",
)
_COUNT_CUES = ("種類数", "いくつ", "何個", "個数", "count", "how many", "number of")
_QUOTE_PATTERNS = (
re.compile(r"「([^」]{1,64})」"), re.compile(r"『([^』]{1,64})』"),
re.compile(r'“([^”]{1,64})”'), re.compile(r'"([^"\n]{1,64})"'),
re.compile(r"'([^'\n]{1,64})'"),
)
@dataclass(frozen=True)
class KeywordQuestion:
task: str
keywords: tuple[str, ...]
option_keys: tuple[str, ...] = ()
def normalized(value: str) -> str:
return "".join(character for character in unicodedata.normalize("NFKC", value).casefold()
if unicodedata.category(character)[0] in {"L", "N"})
def kana(value: str) -> str:
result = []
for character in unicodedata.normalize("NFKC", value):
codepoint = ord(character)
if 0x30A1 <= codepoint <= 0x30F6:
character = chr(codepoint - 0x60)
if unicodedata.category(character)[0] in {"L", "N"} or character == "ー":
result.append(character.casefold())
return "".join(result)
class JapaneseNormalizer:
def __init__(self) -> None:
self.segmenter = dictionary.Dictionary().create()
@lru_cache(maxsize=16_384)
def __call__(self, value: str) -> tuple[str, str, str, frozenset[str], frozenset[str]]:
reading_pieces = []
lemma_surfaces = []
lemma_readings = []
lemma_surface_tokens = set()
lemma_reading_tokens = set()
for morpheme in self.segmenter.tokenize(value, tokenizer.Tokenizer.SplitMode.A):
reading = morpheme.reading_form()
reading_pieces.append(morpheme.surface() if reading == "*" else reading)
lemma = morpheme.dictionary_form()
if not lemma or lemma == "*":
lemma = morpheme.surface()
lemma_surface = normalized(lemma)
lemma_reading_parts = []
for part in self.segmenter.tokenize(lemma, tokenizer.Tokenizer.SplitMode.A):
part_reading = part.reading_form()
lemma_reading_parts.append(part.surface() if part_reading == "*" else part_reading)
lemma_reading = kana("".join(lemma_reading_parts))
if lemma_surface:
lemma_surfaces.append(lemma_surface)
lemma_surface_tokens.add(lemma_surface)
if lemma_reading:
lemma_readings.append(lemma_reading)
lemma_reading_tokens.add(lemma_reading)
return (
kana("".join(reading_pieces)),
"".join(lemma_surfaces),
"".join(lemma_readings),
frozenset(lemma_surface_tokens),
frozenset(lemma_reading_tokens),
)
_JAPANESE = JapaneseNormalizer()
def partial_similarity(keyword: str, transcript: str) -> float:
pattern, text = normalized(keyword), normalized(transcript)
if not pattern or not text:
return 0.0
if pattern in text:
return 1.0
previous = [0] * (len(text) + 1)
for index, left in enumerate(pattern, 1):
current = [index]
for position, right in enumerate(text, 1):
current.append(min(current[-1] + 1, previous[position] + 1,
previous[position - 1] + (left != right)))
previous = current
return max(0.0, 1.0 - min(previous) / len(pattern))
def _quotes(value: str) -> list[str]:
found = []
for pattern in _QUOTE_PATTERNS:
found.extend(match.strip() for match in pattern.findall(value))
return list(dict.fromkeys(value for value in found if normalized(value)))
def _explicit(question: Mapping[str, Any]) -> KeywordQuestion | None:
task = question.get("jev_task")
if task is None:
return None
if task not in TASKS:
raise ValueError(f"unsupported jev_task: {task}")
raw = question.get("keywords")
if not isinstance(raw, list) or not raw or any(
not isinstance(value, str) or not normalized(value) or len(value) > 64 for value in raw
):
raise ValueError("keyword jev_task requires 1 to 32 nonempty keywords")
keywords = tuple(raw)
if len(keywords) > 32 or len(set(map(normalized, keywords))) != len(keywords):
raise ValueError("keywords must be distinct and at most 32 entries")
pairs = question_options(dict(question))
if task == "keyword_presence" and (question.get("type") != "noul" or len(keywords) != 1):
raise ValueError("keyword_presence requires Noul and exactly one keyword")
if task == "keyword_choice" and (question.get("type") != "choice" or len(keywords) != len(pairs)):
raise ValueError("keyword_choice requires one keyword per Choice option")
if task == "keyword_count" and (question.get("type") != "score" or len(pairs) <= len(keywords)):
raise ValueError("keyword_count requires Score levels from zero through the keyword count")
return KeywordQuestion(task, keywords, tuple(key for key, _ in pairs) if task == "keyword_choice" else ())
def classify_keyword_question(question: Mapping[str, Any]) -> KeywordQuestion | None:
explicit = _explicit(question)
if explicit is not None:
return explicit
instructions = question.get("instructions")
if not isinstance(instructions, str):
return None
folded = unicodedata.normalize("NFKC", instructions).casefold()
if not any(cue in folded for cue in _CUES):
return None
kind = question.get("type")
pairs = question_options(dict(question))
if kind == "noul":
values = _quotes(instructions)
return KeywordQuestion("keyword_presence", (values[0],)) if len(values) == 1 else None
if kind == "choice":
values = []
for key, description in pairs:
quoted = _quotes(description)
if len(quoted) == 1:
values.append(quoted[0])
elif normalized(key) and key.casefold() not in {"true", "false", "yes", "no"}:
values.append(key)
else:
return None
if len(set(map(normalized, values))) == len(values):
return KeywordQuestion("keyword_choice", tuple(values), tuple(key for key, _ in pairs))
return None
if kind == "score" and any(cue in folded for cue in _COUNT_CUES):
values = _quotes(instructions)
if not values and (":" in instructions or ":" in instructions):
tail = re.split(r"[::]", instructions, maxsplit=1)[1]
values = [value.strip() for value in re.split(r"\s*[//,、]\s*", tail)
if normalized(value.strip())]
if values and len(pairs) > len(values) and len(set(map(normalized, values))) == len(values):
return KeywordQuestion("keyword_count", tuple(values))
return None
def _keyword_score(
keyword: str,
transcript: str,
transcript_analysis: tuple[str, str, str, frozenset[str], frozenset[str]],
) -> float:
transcript_reading, lemma_surface, lemma_reading, _, _ = transcript_analysis
query_reading = _JAPANESE(keyword)[0]
return max(
partial_similarity(keyword, transcript),
partial_similarity(query_reading, transcript_reading),
partial_similarity(keyword, lemma_surface),
partial_similarity(query_reading, lemma_reading),
)
def _contains(
keyword: str,
transcript: str,
transcript_analysis: tuple[str, str, str, frozenset[str], frozenset[str]],
) -> tuple[bool, float, bool]:
key, text = normalized(keyword), normalized(transcript)
query_reading = _JAPANESE(keyword)[0]
transcript_reading, _, _, lemma_surface_tokens, lemma_reading_tokens = transcript_analysis
exact = bool(
key and (
key in text or (query_reading and query_reading in transcript_reading)
or key in lemma_surface_tokens
or (query_reading and query_reading in lemma_reading_tokens)
)
)
score = 1.0 if exact else _keyword_score(keyword, transcript, transcript_analysis)
length = max(len(key), len(query_reading))
return exact or (length >= MINIMUM_FUZZY_LENGTH and score >= MATCH_THRESHOLD), score, exact
def _presence_probability(score: float) -> float:
if score >= MATCH_THRESHOLD:
return 0.5 + 0.5 * (score - MATCH_THRESHOLD) / (1.0 - MATCH_THRESHOLD)
return 0.5 * score / MATCH_THRESHOLD
def answer_keyword_question(transcript: str, question: Mapping[str, Any],
route: KeywordQuestion | None = None) -> dict[str, Any]:
route = route or classify_keyword_question(question)
if route is None:
raise ValueError("question is not a supported keyword task")
transcript_analysis = _JAPANESE(transcript)
matches = [_contains(keyword, transcript, transcript_analysis) for keyword in route.keywords]
if route.task == "keyword_presence":
probability = _presence_probability(matches[0][1])
answer = typed_answer(dict(question), [1.0 - probability, probability])
elif route.task == "keyword_choice":
scores = [value[1] for value in matches]
maximum = max(scores)
weights = [math.exp((value - maximum) / 0.10) for value in scores]
answer = typed_answer(dict(question), weights)
else:
count = sum(value[0] for value in matches)
levels = len(question_options(dict(question)))
probabilities = [0.0] * levels
probabilities[min(count, levels - 1)] = 1.0
answer = typed_answer(dict(question), probabilities)
answer["confidence_kind"] = "transcript_match_score_not_calibrated_correctness"
answer["keyword_evidence"] = {
"task": route.task, "threshold": MATCH_THRESHOLD,
"minimum_fuzzy_length": MINIMUM_FUZZY_LENGTH,
"matcher": "surface-reading-lemma",
"matches": [{"keyword": keyword, "matched": matched, "score": score, "exact": exact}
for keyword, (matched, score, exact) in zip(route.keywords, matches)],
}
return answer
def answer_keyword_questions(transcript: str, questions: Mapping[str, Mapping[str, Any]]) -> dict[str, dict[str, Any]]:
result = {}
for name, question in questions.items():
route = classify_keyword_question(question)
if route is None:
raise ValueError(f"{name}: not a supported keyword question")
result[name] = answer_keyword_question(transcript, question, route)
return result