Spaces:
Sleeping
Sleeping
File size: 27,283 Bytes
463f12c 8829ab0 463f12c 97939e8 463f12c 8829ab0 97939e8 8829ab0 463f12c 97939e8 463f12c 8829ab0 6538b83 8829ab0 6538b83 97939e8 6538b83 97939e8 6538b83 97939e8 6538b83 97939e8 6538b83 8829ab0 463f12c 88c1c3c 8829ab0 88c1c3c 8829ab0 88c1c3c 8829ab0 88c1c3c 8829ab0 88c1c3c 8829ab0 88c1c3c 8829ab0 88c1c3c 8829ab0 88c1c3c 8829ab0 88c1c3c 8829ab0 88c1c3c 463f12c 8829ab0 463f12c 8829ab0 88c1c3c 97939e8 8829ab0 88c1c3c 97939e8 88c1c3c 8829ab0 88c1c3c 8829ab0 88c1c3c 8829ab0 88c1c3c 8829ab0 88c1c3c 463f12c 8829ab0 463f12c 8829ab0 463f12c 88c1c3c 8829ab0 88c1c3c 8829ab0 463f12c 88c1c3c 8829ab0 463f12c 88c1c3c 8829ab0 463f12c 88c1c3c 463f12c 8829ab0 463f12c 88c1c3c 8829ab0 463f12c 8829ab0 463f12c 8829ab0 463f12c 8829ab0 463f12c 8829ab0 463f12c 8829ab0 463f12c 6538b83 463f12c 6538b83 463f12c 6538b83 463f12c 8829ab0 88c1c3c 463f12c 6538b83 88c1c3c ed4a95a 8829ab0 ed4a95a 88c1c3c 6538b83 88c1c3c 463f12c 8829ab0 88c1c3c 8829ab0 463f12c 88c1c3c 463f12c 8829ab0 88c1c3c ed4a95a 88c1c3c 463f12c 8829ab0 463f12c 8829ab0 463f12c 97939e8 20da4ec 463f12c 97939e8 463f12c 97939e8 463f12c 8829ab0 463f12c 97939e8 463f12c 8829ab0 463f12c 97939e8 463f12c 8829ab0 97939e8 463f12c 8829ab0 97939e8 8829ab0 97939e8 8829ab0 97939e8 8829ab0 463f12c 8829ab0 463f12c 384ad16 463f12c ed4a95a 463f12c 384ad16 ed4a95a 384ad16 ed4a95a 463f12c ed4a95a 463f12c 384ad16 463f12c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578 579 580 581 582 583 584 585 586 587 588 589 590 591 592 593 594 595 596 597 598 599 600 601 | """
Processor module adapted from static.ipynb for VocalGuard.
Provides process_file(path) which returns a dict with final_risk, label, and details.
Includes ML classifier layer for improved accuracy through training.
"""
import os
import sys
import numpy as np
import librosa
import joblib
import logging
from pathlib import Path
# Configure logging
logging.basicConfig(
level=logging.INFO,
format='%(asctime)s - %(levelname)s - %(message)s',
handlers=[logging.StreamHandler(sys.stdout)]
)
logger = logging.getLogger(__name__)
# Optional imports - fallback to heuristics if not available
try:
import torch
from transformers import WhisperProcessor, WhisperForConditionalGeneration
from sentence_transformers import SentenceTransformer
TORCH_AVAILABLE = True
except Exception:
TORCH_AVAILABLE = False
try:
from sklearn.ensemble import RandomForestClassifier
from sklearn.preprocessing import StandardScaler
SKLEARN_AVAILABLE = True
except Exception:
SKLEARN_AVAILABLE = False
# Global model cache (lazy loading)
_whisper_model = None
_whisper_processor = None
_sentence_model = None
_ml_classifier = None
_feature_scaler = None
def get_whisper_model():
"""Lazy load Whisper model"""
global _whisper_model, _whisper_processor
if _whisper_model is None and TORCH_AVAILABLE:
logger.info("Loading Whisper model (first time only)...")
_whisper_processor = WhisperProcessor.from_pretrained(
"openai/whisper-small",
cache_dir=MODEL_DIR
)
_whisper_model = WhisperForConditionalGeneration.from_pretrained(
"openai/whisper-small",
cache_dir=MODEL_DIR,
torch_dtype=torch.float16 if torch.cuda.is_available() else torch.float32,
low_cpu_mem_usage=True
)
logger.info("Whisper model loaded")
return _whisper_processor, _whisper_model
def get_sentence_model():
"""Lazy load sentence transformer"""
global _sentence_model
if _sentence_model is None and TORCH_AVAILABLE:
logger.info("Loading sentence transformer...")
_sentence_model = SentenceTransformer(
'all-MiniLM-L6-v2',
cache_folder=MODEL_DIR
)
logger.info("Sentence model loaded")
return _sentence_model
def get_ml_classifier():
"""Lazy load trained ML classifier if available"""
global _ml_classifier, _feature_scaler
if _ml_classifier is None and SKLEARN_AVAILABLE:
classifier_path = os.path.join(MODEL_DIR, "scam_classifier.pkl")
scaler_path = os.path.join(MODEL_DIR, "feature_scaler.pkl")
if os.path.exists(classifier_path) and os.path.exists(scaler_path):
try:
_ml_classifier = joblib.load(classifier_path)
_feature_scaler = joblib.load(scaler_path)
print("✅ Trained ML classifier loaded - using enhanced accuracy mode")
return _ml_classifier, _feature_scaler
except Exception as e:
print(f"Warning: Could not load ML classifier: {e}")
return None, None
else:
print("ℹ️ No trained classifier found - using rule-based scoring")
print(" To train a classifier, run: python train_model.py")
return _ml_classifier, _feature_scaler
# Config
CHUNK_SECONDS = 10
TARGET_SR = 16000
MODEL_DIR = os.path.join(os.path.dirname(__file__), "models")
os.makedirs(MODEL_DIR, exist_ok=True)
# Comprehensive scam vocabulary - organized by threat category
SCAM_TERMS = {
# Critical urgency words (highest weight)
"urgency": ["urgent", "immediately", "now", "today", "right now", "asap", "expire", "expires",
"expired", "deadline", "final", "last chance", "limited time", "act now", "hurry",
"within hours", "time sensitive", "critical", "emergency", "instant"],
# Authority impersonation (very high weight)
"authority": ["irs", "police", "fbi", "government", "officer", "agent", "department", "federal",
"sheriff", "marshal", "investigator", "prosecutor", "attorney", "social security",
"medicare", "medicaid", "customs", "immigration", "enforcement", "treasury",
"homeland security", "border patrol", "detective", "inspector", "official",
"badge number", "case number", "file number", "claim number"],
# Financial institutions
"financial": ["bank", "credit card", "visa", "mastercard", "paypal", "venmo", "zelle",
"account", "routing number", "card number", "cvv", "pin", "password",
"wells fargo", "chase", "bank of america", "citibank", "capital one",
"transaction", "pending", "overdrawn", "overdraft", "declined"],
# Threat words (high weight)
"threats": ["arrest", "arrested", "warrant", "suspended", "blocked", "frozen", "seized",
"lawsuit", "legal action", "court", "jail", "prison", "penalty", "fine",
"consequences", "investigation", "fraud", "charges", "prosecution", "terminate",
"revoked", "cancelled", "legal trouble", "criminal", "indictment", "subpoena"],
# Information requests (medium-high weight)
"requests": ["verify", "confirm", "provide", "give", "send", "transfer", "wire", "payment",
"social security number", "ssn", "date of birth", "mother maiden", "otp",
"verification code", "access code", "passcode", "authenticate", "validate",
"full name", "address", "zip code", "last four", "account number",
"verification", "proceed", "process", "documentation", "legitimacy",
"protocol", "compliance", "specifics", "appropriate channels"],
# Payment methods often used in scams
"payment": ["gift card", "google play", "itunes", "amazon card", "prepaid card", "reload",
"bitcoin", "cryptocurrency", "crypto", "western union", "money gram", "cash app",
"target card", "walmart card", "steam card", "vanilla visa", "greendot",
"cash pickup", "money order", "cashiers check"],
# Deception/reward words (medium weight)
"deception": ["congratulations", "winner", "won", "prize", "lottery", "sweepstakes", "refund",
"rebate", "claim", "eligible", "selected", "qualified", "free", "guarantee",
"reward", "bonus", "approved", "pre-approved", "exclusive", "special offer",
"limited offer", "one time", "cant miss", "incredible deal", "claims",
"vague", "refuses", "seeks", "maintains", "investment opportunity"],
# Technical support scams
"tech": ["virus", "malware", "hacked", "breach", "compromised", "microsoft", "apple",
"tech support", "computer", "windows", "error", "warning", "firewall",
"security alert", "suspicious activity", "ip address", "remote access",
"teamviewer", "anydesk", "license expired", "subscription", "access",
"security breach", "system alert"],
# Pressure tactics
"pressure": ["must", "need to", "have to", "required", "mandatory", "cannot", "will not",
"unless", "or else", "final notice", "last warning", "do not ignore",
"do not hang up", "stay on line", "dont delay", "no choice", "only option",
"pressures", "before", "directly", "without", "taking action", "attempts to"],
# Isolation tactics (new category)
"isolation": ["do not tell", "keep confidential", "dont share", "between us", "secret",
"do not contact", "handle personally", "direct line", "callback number",
"do not call back", "use this number"],
# Call spoofing indicators (new category)
"spoofing": ["callback", "reference number", "confirmation code", "direct extension",
"secure line", "private line", "department line", "this number only"]
}
# Contextual scam phrases (multi-word patterns)
SCAM_PHRASES = [
"your account has been",
"suspicious activity on your",
"we need to verify",
"call us back at",
"press 1 to",
"final attempt to reach",
"avoid legal action",
"going to be arrested",
"warrant for your arrest",
"social security number has been",
"do not hang up",
"stay on the line",
"gift cards",
"before end of business",
"within 24 hours",
"call back immediately",
"refund is pending",
"account will be closed",
"freeze your account",
# New high-confidence phrases from transcript analysis
"attempts to",
"verification of your",
"security verification",
"proceed with",
"taking action",
"through official channels",
"without verification",
"claims to be",
"refuses to provide",
"maintains urgency",
"directly to avoid"
]
def scam_features(text):
"""Extract advanced rule-based scam features with contextual analysis"""
text_lower = text.lower()
words = text_lower.split()
sentences = [s.strip() for s in text.split('.') if s.strip()]
if len(words) == 0:
logger.debug("Empty text, returning zero features")
return [0.0] * 15
logger.debug(f"Analyzing {len(words)} words, {len(sentences)} sentences")
# Category-weighted scoring
urgency_score = sum(1 for w in words if any(term in text_lower for term in SCAM_TERMS["urgency"])) / len(words)
authority_score = sum(1 for w in words if any(term in text_lower for term in SCAM_TERMS["authority"])) / len(words)
threat_score = sum(1 for w in words if any(term in text_lower for term in SCAM_TERMS["threats"])) / len(words)
request_score = sum(1 for w in words if any(term in text_lower for term in SCAM_TERMS["requests"])) / len(words)
payment_score = sum(1 for w in words if any(term in text_lower for term in SCAM_TERMS["payment"])) / len(words)
isolation_score = sum(1 for w in words if any(term in text_lower for term in SCAM_TERMS["isolation"])) / len(words)
# Contextual phrase detection (more accurate than individual words)
phrase_matches = sum(1 for phrase in SCAM_PHRASES if phrase in text_lower)
phrase_score = min(1.0, phrase_matches / 3.0) # Normalize to 0-1
# Advanced linguistic patterns
question_ratio = text.count('?') / (len(sentences) + 1) # Scammers ask many questions
imperative_ratio = sum(1 for s in sentences if s.strip().startswith(("verify", "confirm", "call", "provide", "send", "press", "do not"))) / (len(sentences) + 1)
number_count = sum(1 for w in words if any(c.isdigit() for c in w)) / len(words) # Phone numbers, account numbers
# Readability (scammers use simple language to manipulate)
avg_word_length = sum(len(w) for w in words) / len(words) if words else 0
readability_score = 1.0 if avg_word_length < 4.5 else 0.0 # Suspiciously simple
# Repetition detection (scammers repeat key points)
unique_words = len(set(words))
repetition_score = 1.0 - (unique_words / len(words)) if words else 0.0
# Combination flags (highly suspicious patterns)
urgency_plus_authority = 1.0 if urgency_score > 0.015 and authority_score > 0.015 else 0.0
threat_plus_request = 1.0 if threat_score > 0.015 and request_score > 0.015 else 0.0
isolation_plus_payment = 1.0 if isolation_score > 0.01 and payment_score > 0.01 else 0.0
return [
urgency_score * 10, # 0: Urgency indicators
authority_score * 10, # 1: Authority impersonation
threat_score * 10, # 2: Threats
request_score * 10, # 3: Information requests
payment_score * 10, # 4: Payment methods
phrase_score * 3, # 5: Contextual phrases (NEW)
question_ratio * 2, # 6: Question density
imperative_ratio * 2, # 7: Command density
number_count * 5, # 8: Number frequency
isolation_score * 8, # 9: Isolation tactics (NEW)
readability_score * 2, # 10: Suspiciously simple (NEW)
repetition_score * 3, # 11: Repetition (NEW)
urgency_plus_authority, # 12: Combo flag 1
threat_plus_request, # 13: Combo flag 2
isolation_plus_payment # 14: Combo flag 3 (NEW)
]
def extract_audio_features_from_array(y, sr=TARGET_SR):
"""Extract audio features for EMS/VAS scoring with temporal dynamics"""
y = y.astype(np.float32)
if len(y) < sr: # Less than 1 second
return np.zeros(14, dtype=np.float32)
# Pitch features (emotional stress indicators)
try:
pitch = librosa.yin(y, fmin=80, fmax=300, sr=sr)
pitch_var = float(np.nanvar(pitch))
pitch_mean = float(np.nanmean(pitch))
# Pitch range (larger range = emotional manipulation)
pitch_range = float(np.nanmax(pitch) - np.nanmin(pitch)) if len(pitch) > 0 else 0.0
except Exception:
pitch_var = 0.0
pitch_mean = 150.0
pitch_range = 0.0
# Energy features (pressure/aggression)
energy = librosa.feature.rms(y=y)
energy_mean = float(np.mean(energy)) if energy.size else 0.0
energy_var = float(np.var(energy)) if energy.size else 0.0
# Energy dynamics (sudden changes = stress/pressure)
energy_delta = float(np.mean(np.abs(np.diff(energy))) if energy.size > 1 else 0.0)
# Speaking rate (scammers often speak fast under pressure)
try:
onset_env = librosa.onset.onset_strength(y=y, sr=sr)
tempo = librosa.beat.tempo(onset_envelope=onset_env, sr=sr)[0]
speaking_rate = float(tempo / 120.0) # Normalize to typical speech
except Exception:
speaking_rate = 1.0
# MFCC features (voice quality and biometrics)
mfcc = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=13)
mfcc_var = float(np.var(mfcc)) if mfcc.size else 0.0
# MFCC delta (voice quality changes over time)
mfcc_delta = float(np.mean(np.abs(librosa.feature.delta(mfcc)))) if mfcc.size > 0 else 0.0
# Spectral features (voice naturalness)
spec_flatness = float(np.mean(librosa.feature.spectral_flatness(y=y))) if y.size else 0.0
spec_centroid = float(np.mean(librosa.feature.spectral_centroid(y=y, sr=sr))) if y.size else 0.0
zcr = float(np.mean(librosa.feature.zero_crossing_rate(y))) if y.size else 0.0
# Harmonic analysis (synthetic voice detection)
harmonic = librosa.effects.harmonic(y)
noise = y - harmonic
harmonic_ratio = float(np.sum(harmonic**2) / (np.sum(noise**2) + 1e-6))
# Jitter and shimmer approximations (voice stability)
# Higher jitter/shimmer = unnatural voice or stress
frame_length = min(2048, len(y))
frames = librosa.util.frame(y, frame_length=frame_length, hop_length=frame_length//2)
frame_energies = np.sqrt(np.mean(frames**2, axis=0))
shimmer = float(np.std(frame_energies) / (np.mean(frame_energies) + 1e-6) if len(frame_energies) > 1 else 0.0)
return np.array([pitch_var, pitch_mean, pitch_range, energy_mean, energy_var, energy_delta,
speaking_rate, mfcc_var, mfcc_delta, spec_flatness, spec_centroid,
zcr, harmonic_ratio, shimmer], dtype=np.float32)
def transcribe_chunk(y, sr=TARGET_SR):
"""Transcribe audio chunk to text"""
y = y.astype(np.float32)
whisper_processor, whisper_model = get_whisper_model()
if whisper_processor is None or whisper_model is None:
return ""
import torch
inputs = whisper_processor(y, sampling_rate=sr, return_tensors="pt")
with torch.no_grad():
ids = whisper_model.generate(inputs.input_features, max_new_tokens=128)
text = whisper_processor.batch_decode(ids, skip_special_tokens=True)[0].lower()
return text
def compute_lts_score_from_text(text):
"""Compute Linguistic Threat Score with contextual feature weighting"""
if not text or len(text.strip()) < 3:
return 0.0
sentence_model = get_sentence_model()
features = np.array(scam_features(text), dtype=np.float32)
# Enhanced weighted feature combination (optimized for 15 features)
# Features: [urgency, authority, threat, request, payment, phrases, questions, imperatives,
# numbers, isolation, readability, repetition, combo1, combo2, combo3]
# Boosted weights for high-confidence indicators: authority, threats, payment, phrases
weights = np.array([0.14, 0.18, 0.16, 0.11, 0.10, 0.14, 0.03, 0.03,
0.02, 0.04, 0.01, 0.01, 0.02, 0.01, 0.00], dtype=np.float32)
rule_score = float(np.clip(np.sum(features * weights), 0.0, 1.0))
if sentence_model is None:
return rule_score
# Semantic embedding analysis with better normalization
try:
emb = sentence_model.encode([text], convert_to_numpy=True, show_progress_bar=False)
emb_norm = np.linalg.norm(emb)
emb_score = min(1.0, emb_norm / 12.0)
except Exception:
emb_score = 0.0
# Higher weight on rule-based (more reliable and interpretable)
final_score = 0.80 * rule_score + 0.20 * emb_score
return float(np.clip(final_score, 0.0, 1.0))
def compute_ems_score_from_audio_features(ems_features_row):
"""Compute Emotional Manipulation Score with temporal dynamics"""
# Features: [pitch_var, pitch_mean, pitch_range, energy_mean, energy_var, energy_delta, speaking_rate]
vals = np.array(ems_features_row, dtype=np.float32)
# More sensitive normalization for better detection
pitch_var_norm = np.clip(vals[0] / 3000.0, 0, 1) # Lowered threshold for better sensitivity
pitch_high = 1.0 if vals[1] > 160 else (0.5 if vals[1] > 140 else 0.0) # Graduated scale
pitch_range_norm = np.clip(vals[2] / 120.0, 0, 1) # Lower threshold
energy_norm = np.clip(vals[3] / 0.2, 0, 1) # More sensitive to volume
energy_var_norm = np.clip(vals[4] / 0.03, 0, 1) # More sensitive to variance
energy_change = np.clip(vals[5] / 0.08, 0, 1) # More sensitive to sudden changes
fast_speech = 1.0 if vals[6] > 1.2 else (0.5 if vals[6] > 1.05 else 0.0) # Graduated scale
# Enhanced weighting emphasizing stress indicators
score = (0.22 * pitch_var_norm + # Increased - key stress indicator
0.13 * pitch_high +
0.17 * pitch_range_norm + # Increased - emotional manipulation
0.18 * energy_norm +
0.13 * energy_var_norm +
0.12 * energy_change + # Increased - pressure tactics
0.05 * fast_speech)
return float(np.clip(score, 0.0, 1.0))
def compute_vas_score(X_vas_raw_row):
"""Compute Voice Authenticity Score with biometric features"""
# Features: [mfcc_var, mfcc_delta, spec_flatness, spec_centroid, zcr, harmonic_ratio, shimmer]
# Higher scores indicate unnatural/synthetic voice
return float(np.clip(
0.22 * X_vas_raw_row[0] + # MFCC variance
0.18 * X_vas_raw_row[1] + # MFCC delta (voice stability)
0.20 * X_vas_raw_row[2] + # Spectral flatness (naturalness)
0.10 * (1 - min(1, X_vas_raw_row[3] / 3000.0)) + # Spectral centroid (inverted)
0.12 * X_vas_raw_row[4] + # Zero crossing rate
0.10 * (1 - min(1, X_vas_raw_row[5] / 10.0)) + # Harmonic ratio (inverted)
0.08 * min(1, X_vas_raw_row[6] * 10), # Shimmer (voice stability)
0.0, 1.0
))
def chunk_audio_file(path, chunk_seconds=CHUNK_SECONDS, sr=TARGET_SR):
"""Split audio file into chunks"""
try:
# Try loading directly
y, sr_loaded = librosa.load(path, sr=sr, mono=True)
except Exception as e:
# If it fails (e.g., WebM format), try using soundfile or audioread
import soundfile as sf
try:
# Try with soundfile first
y, sr_loaded = sf.read(path)
if len(y.shape) > 1:
y = y.mean(axis=1) # Convert to mono
# Resample if needed
if sr_loaded != sr:
import scipy.signal
y = scipy.signal.resample(y, int(len(y) * sr / sr_loaded))
except Exception:
# Last resort: use audioread
import audioread
with audioread.audio_open(path) as f:
sr_loaded = f.samplerate
y = []
for buf in f:
y.append(np.frombuffer(buf, dtype=np.int16).astype(np.float32) / 32768.0)
y = np.concatenate(y)
if len(y.shape) > 1:
y = y.mean(axis=1)
total_len = y.shape[0]
chunk_len = int(chunk_seconds * sr)
chunks = []
for start in range(0, total_len, chunk_len):
end = min(start + chunk_len, total_len)
chunks.append(y[start:end])
return chunks
def process_and_score_chunk(y_chunk, chunk_index):
"""Process a single audio chunk and return scores"""
logger.info(f"[Chunk {chunk_index}] Processing {len(y_chunk)} samples ({len(y_chunk)/TARGET_SR:.2f}s)")
# Transcription
transcription = transcribe_chunk(y_chunk, TARGET_SR)
logger.info(f"[Chunk {chunk_index}] Transcription ({len(transcription)} chars): '{transcription}'")
# LTS
lts = compute_lts_score_from_text(transcription)
logger.info(f"[Chunk {chunk_index}] LTS: {lts:.3f}")
# Audio features (14 features now)
features = extract_audio_features_from_array(y_chunk, TARGET_SR)
logger.info(f"[Chunk {chunk_index}] Audio features: pitch_var={features[0]:.2f}, pitch_mean={features[1]:.2f}, energy_mean={features[3]:.4f}")
# EMS features: pitch_var, pitch_mean, pitch_range, energy_mean, energy_var, energy_delta, speaking_rate
ems_row = features[[0, 1, 2, 3, 4, 5, 6]]
ems = compute_ems_score_from_audio_features(ems_row)
logger.info(f"[Chunk {chunk_index}] EMS: {ems:.3f} (from features: {ems_row})")
# VAS features: mfcc_var, mfcc_delta, spec_flatness, spec_centroid, zcr, harmonic_ratio, shimmer
vas_raw = features[[7, 8, 9, 10, 11, 12, 13]]
# Normalize VAS features (already in reasonable ranges)
vas = compute_vas_score(vas_raw)
logger.info(f"[Chunk {chunk_index}] VAS: {vas:.3f}")
# Base risk calculation (rule-based)
base_risk = 0.5 * lts + 0.3 * ems + 0.2 * vas
logger.info(f"[Chunk {chunk_index}] Base risk: {base_risk:.3f} (LTS={lts:.3f}, EMS={ems:.3f}, VAS={vas:.3f})")
# Try ML classifier for enhanced accuracy
ml_risk = None
ml_classifier, feature_scaler = get_ml_classifier()
if ml_classifier is not None and feature_scaler is not None:
try:
# Combine all features for ML prediction
ml_features = np.concatenate([
np.array(scam_features(transcription)), # 15 linguistic features
features # 14 audio features
]).reshape(1, -1) # Shape: (1, 29)
# Scale features
ml_features_scaled = feature_scaler.transform(ml_features)
# Get probability prediction
ml_proba = ml_classifier.predict_proba(ml_features_scaled)[0]
ml_risk = float(ml_proba[1]) # Probability of scam class
# Ensemble: weighted average of rule-based and ML
final_risk = 0.4 * base_risk + 0.6 * ml_risk
logger.info(f"[Chunk {chunk_index}] ML risk: {ml_risk:.3f}, Final (ensemble): {final_risk:.3f}")
except Exception as e:
logger.warning(f"[Chunk {chunk_index}] ML prediction failed: {e}, using base risk")
final_risk = base_risk
else:
final_risk = base_risk
return {
"chunk_index": int(chunk_index),
"transcription": transcription,
"LTS": float(lts),
"EMS": float(ems),
"VAS": float(vas),
"FINAL_RISK": float(final_risk),
"ml_enhanced": ml_risk is not None
}
def process_file(path):
"""Main entry point: process audio file and return risk assessment"""
chunks = chunk_audio_file(path)
results = []
for idx, c in enumerate(chunks):
try:
r = process_and_score_chunk(c, idx)
except Exception as e:
r = {"chunk_index": idx, "error": str(e), "FINAL_RISK": 0.0, "LTS": 0.0, "EMS": 0.0, "VAS": 0.0}
results.append(r)
# Weighted aggregation: prioritize longer chunks and high-risk detections
final_risks = [r.get("FINAL_RISK", 0.0) for r in results]
lts_scores = [r.get("LTS", 0.0) for r in results if "LTS" in r]
ems_scores = [r.get("EMS", 0.0) for r in results if "EMS" in r]
vas_scores = [r.get("VAS", 0.0) for r in results if "VAS" in r]
# Weight chunks by their audio length (longer = more reliable)
chunk_weights = []
for r in results:
# Longer chunks get higher weight (normalized to 1.0 for 10s chunks)
transcription_len = len(r.get("transcription", ""))
# Weight by transcription length (more content = more reliable)
weight = min(1.0, transcription_len / 100.0) if transcription_len > 0 else 0.1
chunk_weights.append(weight)
# Normalize weights
total_weight = sum(chunk_weights) if sum(chunk_weights) > 0 else 1.0
chunk_weights = [w / total_weight for w in chunk_weights]
# Weighted average with emphasis on high scores (use max of weighted avg and top 2 chunks avg)
if final_risks:
weighted_risk = sum(r * w for r, w in zip(final_risks, chunk_weights))
# Also consider top 2 highest risks (catches strong scam signals)
top_risks = sorted(final_risks, reverse=True)[:2]
top_avg = np.mean(top_risks) if len(top_risks) > 0 else 0.0
# Take the higher of weighted average or top chunks average
overall = float(max(weighted_risk, top_avg * 0.85)) # Slight discount for top avg
else:
overall = 0.0
# Weighted averages for breakdown scores
if lts_scores and chunk_weights:
avg_lts = float(sum(l * w for l, w in zip(lts_scores, chunk_weights[:len(lts_scores)])))
else:
avg_lts = 0.0
if ems_scores and chunk_weights:
avg_ems = float(sum(e * w for e, w in zip(ems_scores, chunk_weights[:len(ems_scores)])))
else:
avg_ems = 0.0
if vas_scores and chunk_weights:
avg_vas = float(sum(v * w for v, w in zip(vas_scores, chunk_weights[:len(vas_scores)])))
else:
avg_vas = 0.0
# Determine label based on improved thresholds (more sensitive)
if overall > 0.60: # Lowered from 0.70
label = "risky"
status = "HIGH RISK"
elif overall > 0.35: # Lowered from 0.40
label = "suspicious"
status = "SUSPICIOUS"
else:
label = "safe"
status = "SAFE"
return {
"final_risk": overall,
"label": label,
"status": status,
"linguistic_threat_score": avg_lts,
"emotional_manipulation_score": avg_ems,
"voice_authenticity_score": avg_vas,
"chunks": results,
"num_chunks": len(results)
}
|