autoace-audio-evaluator / tests /test_analysis.py
mritunjayk-ops's picture
Add LLM emotion classifier, result summary, and honest docs
f94ee13
Raw
History Blame Contribute Delete
13.6 kB
from array import array
from dataclasses import replace
import json
import numpy as np
from autoace_audio.audio import (
AudioMeasurements,
AudioMetadata,
ChannelAnalysis,
NormalizedAudio,
SpeechRegion,
measure_audio,
refine_primary_speech_measurements,
)
from autoace_audio.fusion import fuse_prediction
from autoace_audio.models import (
AcousticEmotionResult,
EnvironmentalSoundResult,
RoleAnalysisResult,
SemanticEmotionResult,
TranscriptResult,
TranscriptSegment,
_excessive_repetition,
grounded_semantic_emotion,
resolve_speaker_roles,
transcript_subset,
)
from autoace_audio.pipeline import AnalysisResult
def normalized(samples, sample_rate=16_000):
metadata = AudioMetadata(
name="synthetic.wav",
container_format="wav",
codec_name="pcm_s16le",
sample_rate=sample_rate,
channels=1,
channel_layout=None,
duration_seconds=len(samples) / sample_rate,
size_bytes=max(len(samples) * 2, 1),
bit_rate=None,
)
return NormalizedAudio(
metadata=metadata,
sample_rate=sample_rate,
samples=array("f", samples),
original_channels=1,
channel_analysis=ChannelAnalysis(False, None, None),
peak_before_normalization=max((abs(float(value)) for value in samples), default=0.0),
normalization_gain=1.0,
)
def base_measurements(**overrides):
values = {
"speech_regions": (SpeechRegion(0.0, 2.0, -20.0),),
"speech_fraction": 0.5,
"longest_internal_silence_seconds": 0.0,
"long_silence_present": False,
"clipping_ratio": 0.0,
"loudness_dbfs": -20.0,
"low_volume": False,
"snr_proxy_db": 25.0,
"spectral_flatness": 0.05,
"static_score": 0.02,
"bandwidth_hz": 6_000.0,
"high_frequency_ratio": 0.2,
"muffled": False,
"noise_present": False,
"noise_severity": "none",
"deterministic_noise_type": "",
"audio_quality": "clear",
"speech_energy_variability_db": 4.0,
"overlap_score": 0.1,
"speaker_overlap_present": False,
"vad_threshold_dbfs": -40.0,
}
values.update(overrides)
return AudioMeasurements(**values)
def test_vad_and_long_internal_silence_detection():
sample_rate = 16_000
time = np.arange(sample_rate, dtype=np.float32) / sample_rate
speech = 0.25 * np.sin(2 * np.pi * 220 * time)
waveform = np.concatenate([speech, np.zeros(sample_rate * 11, dtype=np.float32), speech])
result = measure_audio(normalized(waveform))
assert len(result.speech_regions) == 2
assert result.long_silence_present is True
assert result.longest_internal_silence_seconds >= 10.9
def test_clipping_and_low_volume_are_measured_not_defaulted():
clipped = np.tile(np.array([1.0, -1.0, 0.9, -0.9], dtype=np.float32), 8_000)
quiet_time = np.arange(16_000, dtype=np.float32) / 16_000
quiet = 0.005 * np.sin(2 * np.pi * 440 * quiet_time)
clipped_result = measure_audio(normalized(clipped))
quiet_result = measure_audio(normalized(quiet))
assert clipped_result.clipping_ratio > 0.4
assert clipped_result.audio_quality == "severely_impaired"
assert quiet_result.low_volume is True
assert quiet_result.audio_quality != "clear"
def test_asr_repetition_guard():
assert _excessive_repetition("hello hello hello hello hello") is True
assert _excessive_repetition("the vehicle is ready for pickup tomorrow") is False
def test_semantic_evidence_is_literal_and_asr_gated():
accepted = TranscriptSegment(0.0, 2.0, "Thank you, that sounds good.", 0.8, True, None)
transcript = TranscriptResult(
text=accepted.text,
segments=(accepted,),
confidence=0.8,
reliable=True,
accepted_count=1,
rejected_count=0,
status="ok",
)
result = grounded_semantic_emotion(transcript)
assert result.tone == "satisfied"
assert result.grounded is True
assert all(piece.casefold() in transcript.text.casefold() for piece in result.evidence)
unreliable = TranscriptResult(transcript.text, (accepted,), 0.1, False, 1, 0, "unreliable")
rejected = grounded_semantic_emotion(unreliable)
assert rejected.tone is None
assert rejected.status == "asr_inadequate"
def test_fusion_penalizes_disagreement_and_preserves_noise_invariant():
base = AudioMeasurements(
speech_regions=(),
speech_fraction=0.5,
longest_internal_silence_seconds=0.0,
long_silence_present=False,
clipping_ratio=0.0,
loudness_dbfs=-20.0,
low_volume=False,
snr_proxy_db=25.0,
spectral_flatness=0.05,
static_score=0.02,
bandwidth_hz=6_000.0,
high_frequency_ratio=0.2,
muffled=False,
noise_present=False,
noise_severity="none",
deterministic_noise_type="",
audio_quality="clear",
speech_energy_variability_db=4.0,
overlap_score=0.1,
speaker_overlap_present=False,
vad_threshold_dbfs=-40.0,
)
acoustic = AcousticEmotionResult("neutral", "low", 0.8, "neu", {"neu": 0.8}, 2)
semantic = SemanticEmotionResult("upset", "medium", 0.7, ("unacceptable",), True, "ok")
acoustic_only = fuse_prediction(base, acoustic)
disagreement = fuse_prediction(base, acoustic, semantic, hybrid_requested=True)
assert disagreement.prediction.confidence < acoustic_only.prediction.confidence
assert disagreement.disagreement is True
assert disagreement.prediction.background_noise_type == ""
assert disagreement.prediction.background_noise_severity.value == "none"
def transcript_from(*segments):
confidence = float(np.mean([segment.confidence for segment in segments])) if segments else 0.0
return TranscriptResult(
" ".join(segment.text for segment in segments),
tuple(segments),
confidence,
bool(segments),
len(segments),
0,
"ok" if segments else "unreliable",
)
def test_agent_only_phrases_do_not_determine_customer_emotion():
agent = TranscriptSegment(0.0, 2.0, "Thank you for calling. How can I help?", 0.9, True, None)
transcript = transcript_from(agent)
roles = resolve_speaker_roles(transcript)
assert roles.agent_segments == (agent,)
assert roles.customer_segments == ()
customer_semantic = grounded_semantic_emotion(transcript_subset(transcript, roles.customer_segments))
assert customer_semantic.grounded is False
assert customer_semantic.tone is None
def test_neutral_agent_greeting_cannot_override_upset_customer_segment():
agent = TranscriptSegment(0.0, 1.5, "How can I help?", 0.9, True, None)
customer = TranscriptSegment(2.0, 3.5, "This is unacceptable.", 0.85, True, None)
transcript = transcript_from(agent, customer)
roles = resolve_speaker_roles(transcript)
semantic = grounded_semantic_emotion(transcript_subset(transcript, roles.customer_segments))
acoustic = AcousticEmotionResult(
"upset", "high", 0.72, "ang", {"ang": 0.8}, 1,
persistence=1.0, escalation=0.0, evidence_scope="customer_segments",
)
result = fuse_prediction(
base_measurements(), acoustic, semantic,
hybrid_requested=True, role_analysis=roles, transcript=transcript,
)
assert semantic.tone == "upset"
assert result.prediction.emotional_tone.value == "upset"
def test_repeated_hello_alone_does_not_force_upset():
segments = (
TranscriptSegment(0.0, 1.5, "How can I help?", 0.9, True, None),
TranscriptSegment(4.0, 4.5, "Hello?", 0.7, True, None),
TranscriptSegment(7.0, 7.5, "Hello?", 0.75, True, None),
TranscriptSegment(10.0, 10.5, "Hello?", 0.8, True, None),
)
transcript = transcript_from(*segments)
roles = resolve_speaker_roles(transcript)
acoustic = AcousticEmotionResult(
"neutral", "medium", 0.68, "neu", {"neu": 0.8}, 3,
persistence=1.0, escalation=0.0, evidence_scope="customer_segments",
)
result = fuse_prediction(
base_measurements(), acoustic,
SemanticEmotionResult(None, None, 0.0, (), False, "no_grounded_evidence"),
hybrid_requested=True, role_analysis=roles, transcript=transcript,
)
assert roles.repeated_attention_count == 3
assert result.prediction.emotional_tone.value == "neutral"
def test_repeated_attention_with_persistent_agitated_acoustics_can_increase_severity():
segments = (
TranscriptSegment(0.0, 1.5, "How can I help?", 0.9, True, None),
TranscriptSegment(4.0, 4.5, "Hello?", 0.7, True, None),
TranscriptSegment(7.0, 7.5, "Hello?", 0.75, True, None),
)
transcript = transcript_from(*segments)
roles = resolve_speaker_roles(transcript)
acoustic = AcousticEmotionResult(
"frustrated", "high", 0.68, "ang", {"ang": 0.72}, 2,
persistence=0.8, escalation=0.3, evidence_scope="customer_segments",
)
result = fuse_prediction(
base_measurements(), acoustic,
SemanticEmotionResult(None, None, 0.0, (), False, "no_grounded_evidence"),
hybrid_requested=True, role_analysis=roles, transcript=transcript,
)
assert result.prediction.emotional_tone.value == "upset"
def test_customer_and_agent_text_are_not_mixed_for_semantics():
agent = TranscriptSegment(0.0, 2.0, "Thank you for calling. How can I help?", 0.9, True, None)
customer = TranscriptSegment(3.0, 4.0, "This is unacceptable.", 0.8, True, None)
transcript = transcript_from(agent, customer)
roles = resolve_speaker_roles(transcript)
customer_only = transcript_subset(transcript, roles.customer_segments)
semantic = grounded_semantic_emotion(customer_only)
assert "thank you" not in customer_only.text.casefold()
assert semantic.tone == "upset"
assert all(evidence.casefold() in customer.text.casefold() for evidence in semantic.evidence)
def test_low_role_confidence_reduces_final_confidence():
agent = TranscriptSegment(0.0, 1.0, "How can I help?", 0.9, True, None)
customer = TranscriptSegment(2.0, 4.5, "I need service.", 0.8, True, None)
transcript = transcript_from(agent, customer)
resolved = resolve_speaker_roles(transcript)
high = replace(resolved, role_confidence=0.85)
low = replace(resolved, role_confidence=0.20)
acoustic = AcousticEmotionResult(
"neutral", "low", 0.72, "neu", {"neu": 0.85}, 1,
persistence=1.0, evidence_scope="customer_segments",
)
semantic = SemanticEmotionResult(None, None, 0.0, (), False, "no_grounded_evidence")
high_result = fuse_prediction(
base_measurements(), acoustic, semantic,
hybrid_requested=True, role_analysis=high, transcript=transcript,
)
low_result = fuse_prediction(
base_measurements(), acoustic, semantic,
hybrid_requested=True, role_analysis=low, transcript=transcript,
)
assert low_result.prediction.confidence < high_result.prediction.confidence
def test_background_speech_energy_is_not_automatically_primary_speech_or_overlap():
measurements = base_measurements(
speech_regions=(SpeechRegion(0.0, 10.0, -25.0),),
speech_fraction=1.0,
overlap_score=0.9,
speaker_overlap_present=True,
)
refined = refine_primary_speech_measurements(measurements, 10.0, ((1.0, 2.0),))
assert refined.primary_speech_fraction == 0.1
assert refined.speaker_overlap_present is False
assert refined.overlap_reliability == "indeterminate"
def test_background_television_does_not_automatically_imply_overlap():
measurements = refine_primary_speech_measurements(base_measurements(), 5.0, ((0.5, 2.5),))
acoustic = AcousticEmotionResult(
"neutral", "low", 0.7, "neu", {"neu": 0.8}, 1,
persistence=1.0, evidence_scope="customer_segments",
)
environment = EnvironmentalSoundResult("television", 0.4, ("Television",), "ok")
result = fuse_prediction(measurements, acoustic, environment=environment)
assert result.prediction.background_noise_type == "television"
assert result.prediction.speaker_overlap_present is False
def test_public_download_details_exclude_transcript_and_grounding_text():
customer = TranscriptSegment(1.0, 2.0, "This private phrase is unacceptable.", 0.8, True, None)
transcript = transcript_from(customer)
roles = resolve_speaker_roles(transcript)
semantic = grounded_semantic_emotion(transcript_subset(transcript, roles.customer_segments))
acoustic = AcousticEmotionResult(
"upset", "medium", 0.65, "ang", {"ang": 0.7}, 1,
persistence=1.0, evidence_scope="customer_segments",
)
fused = fuse_prediction(
base_measurements(), acoustic, semantic,
hybrid_requested=True, role_analysis=roles, transcript=transcript,
)
analysis = AnalysisResult(
prediction=fused.prediction,
measurements=base_measurements(),
acoustic=acoustic,
semantic_status=semantic.status,
environment=None,
timings_seconds={"total": 1.0},
fallback_statuses=(),
mode="hybrid",
audio_metadata={},
role_analysis=roles,
_transcript=transcript,
_semantic=semantic,
_fusion=fused,
)
downloadable = json.dumps(analysis.public_details()).casefold()
memory_only = json.dumps(analysis.session_details()).casefold()
assert "this private phrase" not in downloadable
assert "unacceptable" not in downloadable
assert "this private phrase" in memory_only