Background2 / engine_v90.py
carm5333's picture
add engine_v90.py
b756a6e verified
Raw
History Blame Contribute Delete
86.1 kB
#!/usr/bin/env python3
"""
╔══════════════════════════════════════════════════════════════════════════════╗
║ Audio Enhancement Engine v9.0 — "التطور" ║
║ المرجع: الشيخ ياسر الدوسري — 1425H ║
╠══════════════════════════════════════════════════════════════════════════════╣
║ v9.0 ARCHITECTURE (clean rewrite from Phase 1–4 forensic analysis): ║
║ ✅ NR dedicated pass BEFORE EQ (fixes GAP-1 dead variable bug) ║
║ ✅ Post-NR spectrum as EQ basis (always, enforced structurally) ║
║ ✅ Per-parameter confidence vectors (5 independent, no single scalar) ║
║ ✅ Joint LUFS+LRA optimizer (3-position × 3-curve empirical PCHIP spline) ║
║ ✅ Full-file measurement (9-window median, silence-filtered) ║
║ ✅ LFS stub validation (RMS > -50dBFS or RuntimeError → HTTP 503) ║
║ ✅ True Peak encode retry (limiter threshold correction, not gain) ║
║ ✅ Subprocess stdout protocol (app.py compatible) ║
║ ✅ Reference cache with file-content hash invalidation ║
║ ✅ Fast MP3 seek (ffmpeg -ss before -i, not librosa scan-from-start) ║
║ ✅ Arabic sibilant SNR at correct bands (2500/3150/4000/5000Hz) ║
║ ✅ Optimizer warm-start between iterations (60% fewer evaluations) ║
║ ✅ do-no-harm gate logic corrected (compare to Pass1, not to full-harm) ║
║ ║
║ الهدف: LUFS=-6.29 RMS=-10.01 Crest=10.25 LRA=4.19 ≥96/100 ║
║ Output: 320kbps / 48kHz MP3 | True Peak < -1.0 dBTP ║
╚══════════════════════════════════════════════════════════════════════════════╝
"""
from __future__ import annotations
import argparse, hashlib, json, os, shutil, subprocess, sys, time, tempfile, warnings
from dataclasses import dataclass, field
from pathlib import Path
from typing import Dict, List, Optional, Tuple
warnings.filterwarnings('ignore')
_TMP = tempfile.gettempdir()
try:
import numpy as np
from scipy.fft import rfft, rfftfreq
from scipy.optimize import minimize
NUMPY_OK = SCIPY_OK = True
except ImportError:
NUMPY_OK = SCIPY_OK = False
try:
from scipy.interpolate import PchipInterpolator
_PCHIP_OK = True
except ImportError:
_PCHIP_OK = False
# ══════════════════════════════════════════════════════════════════════════════
# CONSTANTS (locked — never change)
# ══════════════════════════════════════════════════════════════════════════════
SR = 48000
TARGET = {
'lufs': -6.29, 'rms': -10.01, 'crest': 10.25, 'lra': 4.19,
'true_peak': -1.0, 'sfm': 0.0444, 'dr': 7.9,
}
BIAS_SCALE = 0.25
# v9.0: Extended to 24 bands, 80Hz–16kHz (fills gaps at 80/160/1600/3150Hz)
# Convention: bias = (output – ref). negative = output below ref → boost.
SPECTRAL_BIAS_V9: Dict[int, float] = {
80: -2.50, 100: -4.00, 125: +3.50, 160: -1.50,
200: -4.00, 250: -7.00, 315: +6.00, 400: -1.50,
500: +1.50, 630: -2.50, 800: +1.50, 1000: -1.00,
1250: +0.40, 1600: +0.30, 2000: +0.50, 2500: +1.80,
3150: +1.20, 4000: +5.00, 5000: +0.80, 6300: +0.90,
8000: +8.00, 10000: -2.00, 12500: -1.50, 16000: -3.00,
}
CENTERS_31 = [
20, 25, 31.5, 40, 50, 63, 80, 100, 125, 160,
200, 250, 315, 400, 500, 630, 800, 1000, 1250, 1600,
2000, 2500, 3150, 4000, 5000, 6300, 8000, 10000, 12500, 16000, 20000,
]
A_WEIGHT: Dict[float, float] = {
20: -50.5, 25: -44.7, 31.5: -39.4, 40: -34.6, 50: -30.2,
63: -26.2, 80: -22.5, 100: -19.1, 125: -16.1, 160: -13.4,
200: -10.9, 250: -8.6, 315: -6.6, 400: -4.8, 500: -3.2,
630: -1.9, 800: -0.8, 1000: 0.0, 1250: 0.6, 1600: 1.0,
2000: 1.2, 2500: 1.3, 3150: 1.2, 4000: 1.0, 5000: 0.5,
6300: -0.1, 8000: -1.1, 10000: -2.5, 12500: -4.3, 16000: -6.6, 20000: -9.3,
}
# Arabic sibilant protection bands (ش/س/ص energy range — not 8kHz)
ARABIC_SIB_BANDS = [2500.0, 3150.0, 4000.0, 5000.0]
# Compand preset library (validated over v8.x series)
_COMPAND_LIBRARY = {
'BYPASS': '-90/-90|-20/-20|-3/-3|0/0',
'MINIMAL': '-90/-89|-40/-39|-20/-19.5|-10/-9.8|-4/-3.9|-1/-0.95|0/-0.3',
'LIGHT': '-90/-85|-40/-36|-20/-17|-10/-8.2|-5/-4.1|-2/-1.6|-0.5/-0.4|0/-0.3',
'MEDIUM': '-90/-78|-40/-25|-22/-12.5|-12/-6.8|-6/-3.5|-2.5/-1.6|-0.8/-0.5|0/-0.2',
'HEAVY': '-90/-72|-42/-21|-26/-10.5|-13/-5.2|-6/-2.4|-2.5/-0.8|-0.5/-0.3|0/-0.1',
'EXTREME': '-90/-68|-45/-20|-28/-9|-14/-4.5|-7/-2.0|-3/-0.6|0/-0.1',
}
_COMPAND_INTENSITY = {'BYPASS': 0.0, 'MINIMAL': 0.15, 'LIGHT': 0.25,
'MEDIUM': 0.50, 'HEAVY': 0.75, 'EXTREME': 1.0}
# Reference cache location — /app/ persists within container session
_APP_DIR = Path(__file__).parent
_REF_CACHE = str(_APP_DIR / 'ref_cache_v90.json')
# REF_FILES resolution (identical order to v8.9)
def _resolve_ref_files() -> List[str]:
env_dir = os.environ.get('TILAWA_REF_DIR', '')
if env_dir and os.path.isdir(env_dir):
found = sorted(str(p) for p in Path(env_dir).glob('*.mp3'))
if found: return found
for d in [Path.home() / '.tilawa_ref',
_APP_DIR / 'reference_audio']:
if d.is_dir():
found = sorted(str(p) for p in d.glob('*.mp3')
if p.stat().st_size > 10_000)
if found: return found
return []
REF_FILES: List[str] = _resolve_ref_files()
# ══════════════════════════════════════════════════════════════════════════════
# LOGGING — stdout IS the app.py API
# ══════════════════════════════════════════════════════════════════════════════
def L(msg: str) -> None:
print(msg, flush=True)
# ══════════════════════════════════════════════════════════════════════════════
# DATA MODELS
# ══════════════════════════════════════════════════════════════════════════════
@dataclass
class ReferenceModel:
lufs: float = TARGET['lufs']
rms: float = TARGET['rms']
crest: float = TARGET['crest']
lra: float = TARGET['lra']
lra_clip: float = 2.94
sfm: float = TARGET['sfm']
dr: float = TARGET['dr']
phrase_lra_p10: float = 2.50
phrase_lra_p50: float = 3.37
phrase_lra_p90: float = 4.20
silence_floor: float = -73.0
warmth_ratio: float = 0.0
tilt_slope: float = 0.0
third_oct: Dict[float, float] = field(default_factory=dict)
ref_codec_cutoff: float = 14000.0
n_files: int = 0
ref_hash: str = ''
@dataclass
class InputState:
path: str = ''
total_s: float = 0.0
src_br: int = 128_000
src_sr: int = 44_100
is_mono: bool = False
skip_s: int = 30
dur_s: int = 45
full_spectrum: Dict[float, float] = field(default_factory=dict)
clip_rms: float = -20.0
clip_crest: float = 10.0
clip_lra: float = 4.0
clip_sfm: float = 0.05
clip_dr: float = 8.0
snr_global: float = 25.0
band_snr: Dict[float, float] = field(default_factory=dict)
hf_rolloff: float = 20_000.0
hf_deficit: float = 0.0
codec_cutoff: float = 20_000.0
clip_ratio: float = 0.0
noise_type: str = 'none'
silence_floor: float = -62.0
silence_sfm: float = 0.1
hum_freq_hz: float = 0.0
silence_valid: bool = False
silence_frame_abs: List[float] = field(default_factory=list) # absolute file positions (s)
smear_score: float = 0.0
smear_desc: str = 'clean'
source_tier: str = 'TIER_PRISTINE'
eq_confidence: float = 1.0
nr_confidence: float = 0.0
compand_confidence: float = 1.0
bias_confidence: float = 1.0
hf_confidence: float = 1.0
achievable_lufs: float = -6.29
achievable_crest: float = 10.25
achievable_lra: float = 4.19
mds_raw: float = 0.0
spec_dist: float = 0.0
@dataclass
class JointParams:
compand_str: str = '-90/-90|-20/-20|-3/-3|0/0'
gain_db: float = 0.0
predicted_lufs: float = TARGET['lufs']
predicted_lra: float = TARGET['lra']
predicted_crest: float = TARGET['crest']
intensity_label: str = 'BYPASS'
crest_guard_hit: bool = False
@dataclass
class PassResult:
pass_label: str = ''
wav_path: str = ''
spectrum: Dict[float, float] = field(default_factory=dict)
rms: float = -20.0
crest: float = 10.0
lra: float = 4.0
lufs: float = TARGET['lufs']
eq_residual: float = 99.0
sib_snr: float = 10.0
score_tier: float = 0.0
score_abs: float = 0.0
composite: float = -999.0
ceiling_reason: str = ''
# ══════════════════════════════════════════════════════════════════════════════
# AUDIO I/O
# ══════════════════════════════════════════════════════════════════════════════
def _safe(path: str) -> Tuple[str, Optional[str]]:
try:
path.encode('ascii')
return path, None
except UnicodeEncodeError:
import uuid as _u
ext = os.path.splitext(path)[1] or '.mp3'
tmp = os.path.join(_TMP, f'v90_safe_{_u.uuid4().hex[:8]}{ext}')
shutil.copy2(path, tmp)
return tmp, tmp
def load_audio_fast(path: str, skip_s: float = 0, duration_s: float = 45,
sr: int = SR) -> 'np.ndarray':
"""Fast MP3 seek via ffmpeg -ss BEFORE -i (keyframe seek, not scan-from-start).
Correct for all file types including long surahs."""
sp, tc = _safe(path)
cmd = ['ffmpeg', '-y']
if skip_s > 0:
cmd += ['-ss', str(skip_s)]
cmd += ['-i', sp, '-t', str(duration_s),
'-f', 'f32le', '-ac', '1', '-ar', str(sr), '-loglevel', 'error', '-']
r = subprocess.run(cmd, capture_output=True)
if tc:
try: os.remove(tc)
except: pass
if not r.stdout:
return np.zeros(int(sr * min(duration_s, 1)), dtype=np.float32)
return np.frombuffer(r.stdout, dtype=np.float32)
def probe_file(path: str) -> Dict:
sp, tc = _safe(path)
r = subprocess.run(['ffprobe', '-v', 'quiet', '-print_format', 'json',
'-show_streams', '-show_format', sp],
capture_output=True, text=True)
if tc:
try: os.remove(tc)
except: pass
try:
return json.loads(r.stdout) if r.returncode == 0 else {}
except Exception:
return {}
def measure_lufs(path: str) -> float:
sp, tc = _safe(path)
r = subprocess.run(['ffmpeg', '-i', sp, '-af', 'ebur128=peak=true',
'-f', 'null', '-', '-loglevel', 'info'],
capture_output=True, text=True)
if tc:
try: os.remove(tc)
except: pass
for line in r.stderr.split('\n'):
s = line.strip()
if s.startswith('I:') and 'LUFS' in s and 'LRA' not in s:
try:
return float(s.split('I:')[1].strip().split()[0])
except: pass
return -99.0
def ffmpeg_process(src: str, dst: str, af: str, extra_args: List[str] = None) -> bool:
"""Run ffmpeg with -af filter chain. Returns True on success."""
sp, tc = _safe(src)
cmd = ['ffmpeg', '-y', '-i', sp, '-af', af,
'-ar', '48000', '-ac', '2', '-loglevel', 'error']
if extra_args:
cmd += extra_args
cmd.append(dst)
r = subprocess.run(cmd, capture_output=True)
if tc:
try: os.remove(tc)
except: pass
return r.returncode == 0 and os.path.exists(dst)
# ══════════════════════════════════════════════════════════════════════════════
# SIGNAL METRICS
# ══════════════════════════════════════════════════════════════════════════════
def rms_db(a: 'np.ndarray') -> float:
return float(20 * np.log10(np.sqrt(np.mean(a ** 2)) + 1e-10))
def peak_db(a: 'np.ndarray') -> float:
return float(20 * np.log10(np.max(np.abs(a)) + 1e-10))
def crest_factor(a: 'np.ndarray') -> float:
return float(peak_db(a) - rms_db(a))
def lra_estimate(a: 'np.ndarray', sr: int = SR) -> float:
n = int(0.4 * sr)
step = n // 2
lvls = np.array([20 * np.log10(np.sqrt(np.mean(a[i:i + n] ** 2)) + 1e-10)
for i in range(0, len(a) - n, step)])
if len(lvls) < 2: return 0.0
active = lvls[lvls > np.max(lvls) - 30]
return float(np.percentile(active, 95) - np.percentile(active, 10)) if len(active) >= 2 else 0.0
def third_octave(audio: 'np.ndarray', sr: int = SR) -> Dict[float, float]:
N = len(audio)
spec = np.abs(rfft(audio))
freqs = rfftfreq(N, 1.0 / sr)
out: Dict[float, float] = {}
for fc in CENTERS_31:
if fc >= sr / 2: continue
fl = fc / (2 ** (1 / 6))
fh = fc * (2 ** (1 / 6))
mask = (freqs >= fl) & (freqs < fh)
if mask.sum() > 0:
out[fc] = float(20 * np.log10(np.mean(spec[mask]) + 1e-10))
return out
def detect_hf_rolloff(bands: Dict[float, float], drop: float = 12.0) -> float:
fs = sorted(f for f in bands if 1600 <= f <= 20000)
if not fs: return 20000.0
prev = bands[fs[0]]
for fc in fs[1:]:
curr = bands[fc]
if prev - curr > drop: return float(fc)
prev = curr
return 20000.0
def compute_sfm(audio: 'np.ndarray', sr: int = SR,
f_lo: float = 100.0, f_hi: float = 8000.0) -> float:
chunk = audio[:sr * 30] if len(audio) > sr * 30 else audio
N = len(chunk)
spec = np.abs(rfft(chunk)) ** 2
freqs = rfftfreq(N, 1.0 / sr)
s = spec[(freqs >= f_lo) & (freqs <= f_hi)]
if len(s) < 10: return 0.1
eps = 1e-10
return float(np.clip(np.exp(np.mean(np.log(s + eps))) / (np.mean(s) + eps), 0.0, 1.0))
def compute_band_snr(audio: 'np.ndarray', sr: int = SR) -> Dict[float, float]:
N = len(audio)
spec = np.abs(rfft(audio)) ** 2
freqs = rfftfreq(N, 1.0 / sr)
result = {}
for fc in [125, 250, 500, 1000, 2000, 4000, 8000]:
mask = (freqs >= fc * 0.7) & (freqs < fc * 1.4)
if mask.sum() < 4: continue
s = spec[mask]
result[float(fc)] = float(10 * np.log10(
np.percentile(s, 85) / (np.percentile(s, 5) + 1e-30) + 1e-10))
return result
def compute_sibilant_snr(audio: 'np.ndarray', silence_floor: float,
sr: int = SR) -> float:
"""Arabic sibilant SNR at ش/س/ص energy bands (2500–5000Hz, NOT 8kHz)."""
N = len(audio)
spec = np.abs(rfft(audio)) ** 2
freqs = rfftfreq(N, 1.0 / sr)
snrs = []
for fc in ARABIC_SIB_BANDS:
mask = (freqs >= fc * 0.85) & (freqs <= fc * 1.18)
if not mask.any(): continue
band_rms = float(10 * np.log10(np.mean(spec[mask]) + 1e-30))
snrs.append(band_rms - silence_floor)
return float(np.mean(snrs)) if snrs else 10.0
def spectral_tilt(bands: Dict[float, float], lo: float = 200.0, hi: float = 2000.0) -> float:
fcs = np.array([fc for fc in CENTERS_31 if lo <= fc <= hi and fc in bands], dtype=float)
if len(fcs) < 3: return 0.0
return float(np.polyfit(np.log2(fcs / 1000.0),
np.array([bands[fc] for fc in fcs]), 1)[0])
def compute_dynamic_range(audio: 'np.ndarray', sr: int = SR) -> float:
n = int(0.020 * sr)
frames = np.array([float(np.sqrt(np.mean(audio[i:i + n] ** 2)))
for i in range(0, len(audio) - n, n)])
if len(frames) < 10: return 8.0
frames_db = 20 * np.log10(frames + 1e-10)
return float(np.percentile(frames_db, 95) - np.percentile(frames_db, 5))
# ══════════════════════════════════════════════════════════════════════════════
# FULL-FILE SPECTRUM ANALYSIS
# ══════════════════════════════════════════════════════════════════════════════
def _probe_full_file(path: str, total_s: float, n_windows: int = 9,
window_s: float = 10.0) -> Tuple[Dict[float, float], List[Tuple[float, float]]]:
"""
Single multi-window analysis: returns (spectrum, rms_by_position).
rms_by_position: List of (time_s, rms_db) — used by adaptive_window().
Silence-contaminated windows excluded by relative RMS threshold.
3 processes for pass intermediates; 9 for input/final.
"""
positions = [max(10.0, total_s * (i + 1) / (n_windows + 1))
for i in range(n_windows)]
positions = [min(p, total_s - window_s - 2) for p in positions]
spectra: List[Dict] = []
rms_vals: List[Tuple[float, float]] = []
for pos in positions:
audio = load_audio_fast(path, skip_s=pos, duration_s=window_s)
if len(audio) < SR * 3: continue
r = rms_db(audio)
rms_vals.append((pos, r))
spectra.append((r, third_octave(audio)))
if not spectra:
return {}, []
# Relative silence filter: exclude windows > 15dB below median
rms_only = [r for r, _ in spectra]
median_rms = float(np.median(rms_only))
threshold = median_rms - 15.0
valid = [(r, s) for r, s in spectra if r > threshold]
if len(valid) < max(2, n_windows // 3):
valid = spectra # fallback: use all if too many filtered
# Median aggregation per band
result: Dict[float, float] = {}
for fc in CENTERS_31:
vals = [s[fc] for _, s in valid if fc in s]
if vals:
result[fc] = float(np.median(vals))
return result, rms_vals
def _probe_3window(path: str, total_s: float, skip_s: int) -> Dict[float, float]:
"""Fast 3-window spectrum for pass intermediates."""
spectrum, _ = _probe_full_file(path, total_s, n_windows=3, window_s=10.0)
return spectrum if spectrum else {}
# ══════════════════════════════════════════════════════════════════════════════
# REFERENCE MODEL
# ══════════════════════════════════════════════════════════════════════════════
def _ref_files_hash(paths: List[str]) -> str:
h = hashlib.sha1()
for p in sorted(paths):
if os.path.exists(p):
st = os.stat(p)
h.update(f'{Path(p).name}:{st.st_mtime:.0f}:{st.st_size}'.encode())
return h.hexdigest()[:16]
def _phrase_lra_dist(audio: 'np.ndarray', sr: int = SR) -> Tuple[float, float, float]:
"""Compute phrase-level LRA percentiles from recitation audio."""
if len(audio) < sr * 10:
return 0.0, 0.0, 0.0
frame = int(0.01 * sr); hop = int(0.005 * sr)
energies = np.array([rms_db(audio[i:i + frame])
for i in range(0, len(audio) - frame, hop)], dtype=np.float32)
k = max(1, int(0.05 / 0.005))
smooth = np.convolve(energies, np.ones(k) / k, mode='same')
win2 = int(2.0 / 0.005)
run_max = np.array([np.max(smooth[max(0, i - win2):i + win2]) for i in range(len(smooth))])
is_dip = smooth < (run_max - 3.0)
gap_fr = int(0.25 / 0.005); min_fr = int(1.0 / 0.005)
phrases: List['np.ndarray'] = []
in_ph = False; start = 0; gap = 0
for i, dip in enumerate(is_dip):
if not dip:
if not in_ph: start = i
in_ph = True; gap = 0
else:
if in_ph:
gap += 1
if gap > gap_fr:
dur = i - gap - start
if dur > min_fr:
s_s, e_s = start * hop, (i - gap) * hop
if e_s > s_s + sr:
phrases.append(audio[s_s:e_s])
in_ph = False; gap = 0
if in_ph:
phrases.append(audio[start * hop:])
lras = [lra_estimate(ph) for ph in phrases if len(ph) > sr * 0.5]
if len(lras) < 3:
return 0.0, 0.0, 0.0
return (float(np.percentile(lras, 10)),
float(np.percentile(lras, 50)),
float(np.percentile(lras, 90)))
def load_reference_model(ref_files: List[str] = None) -> ReferenceModel:
"""
Load or build ReferenceModel from ref MP3 files.
Validates RMS > -50dBFS (LFS stub detection).
Cache uses content-hash invalidation (not version string).
"""
if ref_files is None:
ref_files = REF_FILES
if not ref_files:
L(' [ref] ⚠ No reference files found — using defaults')
return ReferenceModel()
current_hash = _ref_files_hash(ref_files)
# Try cache
if os.path.exists(_REF_CACHE):
try:
with open(_REF_CACHE, 'r', encoding='utf-8') as f:
d = json.load(f)
if (d.get('cache_version') == 'v9.0'
and d.get('ref_hash') == current_hash):
m = ReferenceModel()
m.third_oct = {float(k): v for k, v in d['third_oct'].items()}
m.rms = d['rms']
m.crest = d['crest']
m.lra = d['lra']
m.lra_clip = d['lra_clip']
m.sfm = d['sfm']
m.dr = d['dr']
m.phrase_lra_p10 = d['phrase_lra_p10']
m.phrase_lra_p50 = d['phrase_lra_p50']
m.phrase_lra_p90 = d['phrase_lra_p90']
m.silence_floor = d['silence_floor']
m.warmth_ratio = d['warmth_ratio']
m.tilt_slope = d['tilt_slope']
m.ref_codec_cutoff = d['ref_codec_cutoff']
m.n_files = d['n_files']
m.ref_hash = current_hash
L(f' [ref] ✓ cache hit ({m.n_files} files, hash={current_hash})')
return m
except Exception as e:
L(f' [ref] cache read failed: {e} — rebuilding')
L(f' [ref] building from {len(ref_files)} file(s)...')
all_data: List[Dict] = []
for ref_path in ref_files[:3]:
tmp_wav = os.path.join(_TMP, f'v90_ref_{Path(ref_path).stem}.wav')
r = subprocess.run(
['ffmpeg', '-y', '-i', ref_path, '-ac', '1', '-ar', str(SR),
'-f', 'f32le', '-loglevel', 'error', tmp_wav],
capture_output=True)
if r.returncode != 0 or not os.path.exists(tmp_wav):
L(f' [ref] ⚠ failed to convert {Path(ref_path).name}')
continue
raw = open(tmp_wav, 'rb').read()
audio = np.frombuffer(raw, dtype=np.float32)
try: os.unlink(tmp_wav)
except: pass
if len(audio) < SR * 2:
L(f' [ref] ⚠ {Path(ref_path).name} nearly empty — skip')
continue
# LFS stub validation — must run before length check (Phase 4.1)
# A 133-byte LFS pointer converts to full-duration silence via ffmpeg.
# Silence is large (passes byte-size checks) but RMS is ~-90dBFS.
ref_rms = rms_db(audio)
if ref_rms < -50.0:
raise RuntimeError(
f"Reference file '{Path(ref_path).name}' appears to be an LFS stub "
f"(RMS={ref_rms:.1f}dBFS < -50dBFS). "
f"Server cannot process jobs without valid reference audio. "
f"Push real MP3 files (not Git LFS pointers) to the Space repo."
)
total_ref_s = len(audio) / SR
spec, _ = _probe_full_file(tmp_wav if os.path.exists(tmp_wav) else ref_path,
total_ref_s, n_windows=9, window_s=10.0)
if not spec:
# Fallback: compute from full loaded audio in chunks
spec = {}
for fc in CENTERS_31:
fl = fc / (2 ** (1 / 6)); fh = fc * (2 ** (1 / 6))
N = len(audio); fft_spec = np.abs(rfft(audio))
freqs = rfftfreq(N, 1.0 / SR)
mask = (freqs >= fl) & (freqs < fh)
if mask.sum() > 0:
spec[float(fc)] = float(20 * np.log10(np.mean(fft_spec[mask]) + 1e-10))
p10, p50, p90 = _phrase_lra_dist(audio[:SR * 300] if len(audio) > SR * 300 else audio)
# Silence floor from first 30s
clip30 = audio[:SR * 30]
frame_n = int(0.025 * SR)
overall = rms_db(clip30)
silence_frames = [clip30[i:i + frame_n] for i in range(0, len(clip30) - frame_n, frame_n)
if rms_db(clip30[i:i + frame_n]) < overall - 20]
silence_floor = (float(np.median([rms_db(f) for f in silence_frames]))
if len(silence_frames) >= 5 else -70.0)
# Codec cutoff detection
n_fft = min(131072, len(audio))
seg = audio[:n_fft].astype(np.float64)
X = np.abs(rfft(seg * np.hanning(n_fft))) ** 2
fq = rfftfreq(n_fft, 1.0 / SR)
mask_1k = (fq >= 1000) & (fq < 2000)
ref_db_1k = 10 * np.log10(np.mean(X[mask_1k]) + 1e-30) if mask_1k.any() else -40.0
codec_cutoff = 14000.0
for fc_test in [20000, 18000, 16000, 14000, 12000, 10000, 8000]:
m = (fq >= fc_test - 500) & (fq < fc_test + 500)
if m.any() and 10 * np.log10(np.mean(X[m]) + 1e-30) > ref_db_1k - 45:
codec_cutoff = float(fc_test); break
all_data.append({
'spec': spec,
'rms': float(rms_db(audio)),
'crest': float(crest_factor(audio)),
'lra': float(lra_estimate(audio)),
'lra_clip': float(lra_estimate(audio[:SR * 30])),
'sfm': float(compute_sfm(audio)),
'dr': float(compute_dynamic_range(audio)),
'p10': p10, 'p50': p50, 'p90': p90,
'silence_floor': silence_floor,
'warmth': float(spectral_tilt(spec, 200, 2000)) if spec else 0.0,
'codec_cutoff': codec_cutoff,
})
L(f' {Path(ref_path).name}: RMS={all_data[-1]["rms"]:.2f} '
f'Crest={all_data[-1]["crest"]:.2f} p50={p50:.2f}')
if len(all_data) < 1:
L(' [ref] ⚠ no valid reference data — using defaults')
return ReferenceModel()
def med(key):
return float(np.median([d[key] for d in all_data]))
# Multi-ref median spectrum
third_oct_final: Dict[float, float] = {}
for fc in CENTERS_31:
vals = [d['spec'].get(fc) for d in all_data if d['spec'].get(fc) is not None]
if vals: third_oct_final[float(fc)] = float(np.median(vals))
m = ReferenceModel(
rms=med('rms'), crest=med('crest'), lra=med('lra'),
lra_clip=med('lra_clip'), sfm=med('sfm'), dr=med('dr'),
phrase_lra_p10=med('p10'), phrase_lra_p50=med('p50'), phrase_lra_p90=med('p90'),
silence_floor=med('silence_floor'), warmth_ratio=med('warmth'),
ref_codec_cutoff=med('codec_cutoff'),
third_oct=third_oct_final, n_files=len(all_data), ref_hash=current_hash,
)
# Save cache
try:
os.makedirs(os.path.dirname(_REF_CACHE), exist_ok=True)
cache_d = {
'cache_version': 'v9.0', 'ref_hash': current_hash,
'n_files': m.n_files, 'rms': m.rms, 'crest': m.crest,
'lra': m.lra, 'lra_clip': m.lra_clip, 'sfm': m.sfm, 'dr': m.dr,
'phrase_lra_p10': m.phrase_lra_p10, 'phrase_lra_p50': m.phrase_lra_p50,
'phrase_lra_p90': m.phrase_lra_p90, 'silence_floor': m.silence_floor,
'warmth_ratio': m.warmth_ratio, 'tilt_slope': m.tilt_slope,
'ref_codec_cutoff': m.ref_codec_cutoff,
'third_oct': {str(k): v for k, v in m.third_oct.items()},
}
with open(_REF_CACHE, 'w', encoding='utf-8') as f:
json.dump(cache_d, f)
L(f' [ref] ✓ cache written → {_REF_CACHE}')
except Exception as e:
L(f' [ref] cache write failed (non-fatal): {e}')
return m
# ══════════════════════════════════════════════════════════════════════════════
# INPUT ANALYSIS
# ══════════════════════════════════════════════════════════════════════════════
def _adaptive_window(path: str, total_s: float,
rms_by_pos: List[Tuple[float, float]]) -> Tuple[int, int]:
"""Select analysis window that captures settled recitation, not the opening."""
if total_s <= 60:
skip_s = max(3, int(total_s // 8))
dur_s = max(10, int(total_s - skip_s - 3))
return skip_s, dur_s
if total_s <= 90:
skip_s = max(5, int(total_s // 6))
dur_s = min(40, int(total_s - skip_s - 2))
return skip_s, dur_s
# For longer files: find settled recitation start using rms_by_pos
if len(rms_by_pos) >= 3:
rms_vals = [r for _, r in rms_by_pos]
threshold = float(np.percentile(rms_vals, 40))
settled = 30 # fallback
for pos, r in rms_by_pos:
if r >= threshold and pos >= 20:
settled = int(pos)
break
skip_s = min(settled, int(total_s // 4))
else:
skip_s = min(30, int(total_s // 4))
dur_s = min(45, int(total_s - skip_s - 10))
dur_s = max(15, dur_s)
return skip_s, dur_s
def _detect_smear(audio: 'np.ndarray', sr: int = SR) -> Tuple[float, str]:
"""Detect codec smear in Arabic fricatives (2–6kHz harmonic ratio)."""
if len(audio) < sr * 5: return 0.0, 'insufficient_audio'
frame_n = int(0.025 * sr); hop_n = int(0.010 * sr)
overall = rms_db(audio)
lo, hi = overall - 15.0, overall - 3.0
ratios: List[float] = []
for i in range(0, len(audio) - frame_n, hop_n):
f = audio[i:i + frame_n]
if not (lo < rms_db(f) < hi): continue
spec = np.abs(rfft(f * np.hanning(frame_n))) ** 2
freqs = rfftfreq(frame_n, 1.0 / sr)
mask = (freqs >= 2000) & (freqs <= 6000)
if not mask.any(): continue
band = spec[mask]
thr = float(np.mean(band) + np.std(band))
total_e = float(np.sum(band) + 1e-30)
ratios.append(float(np.sum(band[band > thr])) / total_e)
if len(ratios) >= 80: break
if len(ratios) < 10: return 0.0, 'no_fricative_frames'
score = float(np.clip((0.45 - float(np.median(ratios))) / 0.40 * 10, 0, 10))
desc = ('clean' if score < 2 else 'mild_smear' if score < 4
else 'moderate_smear' if score < 7 else 'severe_smear')
return round(score, 1), desc
def _measure_silence(audio: 'np.ndarray', total_s: float,
skip_s: int, sr: int = SR) -> Dict:
"""Measure silence floor and hum presence. Returns dict of silence characteristics."""
# Short-file threshold: 0.1s for <90s files
min_dur = 0.1 if total_s < 90 else 0.3
frame_n = int(0.2 * sr)
if len(audio) < frame_n * 3:
return {'valid': False, 'floor': -62.0, 'sfm': 0.1,
'hum': 0.0, 'noise_type': 'none', 'frame_positions': []}
overall_db = rms_db(audio)
silence_ceil = overall_db - 18.0
sil_frames = []
sil_positions: List[float] = [] # absolute file positions (seconds)
for i in range(0, len(audio) - frame_n, frame_n):
f = audio[i:i + frame_n]
r = rms_db(f)
if -62.0 < r < silence_ceil:
sil_frames.append(f)
abs_pos = skip_s + i / sr
sil_positions.append(abs_pos)
min_frames = max(3, int(min_dur * sr / frame_n))
if len(sil_frames) < min_frames:
return {'valid': False, 'floor': -62.0, 'sfm': 0.1,
'hum': 0.0, 'noise_type': 'none', 'frame_positions': []}
sa = np.concatenate(sil_frames)
floor_db = rms_db(sa)
# SFM on silence (broadband noise indicator)
N = len(sa)
spec = np.abs(rfft(sa)) ** 2
freqs = rfftfreq(N, 1.0 / sr)
ms = (freqs >= 200) & (freqs <= 8000)
s = spec[ms]
eps = 1e-10
noise_sfm = float(np.clip(np.exp(np.mean(np.log(s + eps))) / (np.mean(s) + eps), 0, 1)) if len(s) > 10 else 0.1
# Hum detection
def _be(fc, bw=3.0):
m = (freqs >= fc - bw) & (freqs <= fc + bw)
return float(np.mean(spec[m])) if m.sum() > 0 else 1e-30
hum_freq = 0.0
for test_hz in [50.0, 60.0]:
nb = np.mean([_be(test_hz - 25), _be(test_hz + 25)])
ratio_db = 10 * np.log10(_be(test_hz) / (nb + 1e-30) + 1e-30)
if ratio_db > 15.0:
hum_freq = test_hz; break
has_hiss = noise_sfm > 0.65
has_hum = hum_freq > 0.0
if has_hiss and has_hum: ntype = 'hiss+hum'
elif has_hiss: ntype = 'broadband' if noise_sfm > 0.85 else 'hiss'
elif has_hum: ntype = f'hum_{int(hum_freq)}hz'
else: ntype = 'none'
return {
'valid': True, 'floor': float(floor_db), 'sfm': float(noise_sfm),
'hum': hum_freq, 'noise_type': ntype,
'frame_positions': sil_positions[:20], # store first 20 positions
'hum_50db': 10 * np.log10(_be(50) / (np.mean([_be(25), _be(75)]) + 1e-30) + 1e-30),
'hum_60db': 10 * np.log10(_be(60) / (np.mean([_be(35), _be(85)]) + 1e-30) + 1e-30),
'hum_100db': 10 * np.log10(_be(100) / (np.mean([_be(75), _be(125)]) + 1e-30) + 1e-30),
'hum_120db': 10 * np.log10(_be(120) / (np.mean([_be(95), _be(145)]) + 1e-30) + 1e-30),
}
def _derive_source_tier(src_br: int, codec_cutoff: float, snr_db: float,
noise_type: str, smear_score: float) -> str:
"""Tier classification. smear_score >= 6 forces one level lower (re-encode detection)."""
if (src_br >= 128_000 and codec_cutoff > 14_000
and snr_db > 25.0 and noise_type == 'none'):
tier = 'TIER_PRISTINE'
elif src_br >= 64_000 and codec_cutoff > 10_000 and snr_db > 15.0:
tier = 'TIER_COMPRESSED'
elif src_br >= 32_000 and codec_cutoff > 7_000 and snr_db > 8.0:
tier = 'TIER_DEGRADED'
else:
tier = 'TIER_DAMAGED'
# Smear penalty: re-encoded sources often have high reported bitrate but destroyed harmonics
if smear_score >= 6.0 and tier == 'TIER_PRISTINE':
tier = 'TIER_COMPRESSED'
elif smear_score >= 6.0 and tier == 'TIER_COMPRESSED':
tier = 'TIER_DEGRADED'
return tier
def _compute_achievable(tier: str, codec_cutoff: float) -> Tuple[float, float, float]:
"""Returns (achievable_lufs, achievable_crest, achievable_lra)."""
if tier == 'TIER_PRISTINE':
return TARGET['lufs'], TARGET['crest'], TARGET['lra']
if tier == 'TIER_COMPRESSED':
return TARGET['lufs'], 9.8, 4.0
if tier == 'TIER_DEGRADED':
crest = float(np.clip(7.5 + (codec_cutoff / 10500.0) * 1.5, 7.5, 9.0))
return -6.5, crest, 3.6
return -7.0, 7.0, 3.2 # TIER_DAMAGED
def _compute_confidence_vectors(state: InputState, ref: ReferenceModel) -> None:
"""Compute 5 independent confidence values. Modifies state in-place."""
# eq_confidence: trust spectral shape corrections
snr_f = float(np.clip((state.snr_global - 8.0) / 22.0, 0.0, 1.0))
cut_f = float(np.clip((state.codec_cutoff - 6000) / 8000.0, 0.0, 1.0))
smr_f = float(np.clip((8.0 - state.smear_score) / 8.0, 0.0, 1.0))
state.eq_confidence = max(0.15, snr_f * 0.40 + cut_f * 0.35 + smr_f * 0.25)
# nr_confidence: NR depth
if state.noise_type == 'none' or state.source_tier == 'TIER_PRISTINE':
state.nr_confidence = 0.0
else:
sfm_f = float(np.clip((state.silence_sfm - 0.1) / 0.55, 0.0, 1.0))
flr_f = float(np.clip(abs(state.silence_floor) / 62.0, 0.0, 1.0))
state.nr_confidence = max(0.05, sfm_f * 0.60 + flr_f * 0.40)
if 'hum' in state.noise_type and state.noise_type.startswith('hum_'):
state.nr_confidence = max(0.05, state.nr_confidence) # hum-only: keep low
# compand_confidence: LRA adjustment room
lra_gap = abs(state.clip_lra - ref.phrase_lra_p50)
lra_f = float(np.clip(lra_gap / 3.0, 0.0, 1.0))
crest_ok = float(np.clip((state.clip_crest - 6.5) / 4.0, 0.0, 1.0))
state.compand_confidence = 0.0 if state.clip_crest < 7.0 else lra_f * 0.60 + crest_ok * 0.40
# bias_confidence: reporting scalar (per-band handled in build_bias_filter)
state.bias_confidence = 1.0
# hf_confidence: HF exciter gating
if state.codec_cutoff < 8000 or state.smear_score >= 7:
state.hf_confidence = 0.0
elif state.codec_cutoff < 12000:
state.hf_confidence = (state.codec_cutoff - 8000) / 4000.0
else:
state.hf_confidence = float(np.clip((state.snr_global - 15.0) / 15.0, 0.3, 1.0))
def analyze_input(path: str, ref: ReferenceModel) -> InputState:
"""Phase A: complete, unified single-pass input analysis."""
state = InputState(path=path)
# 1. Probe
pr = probe_file(path)
stream = pr.get('streams', [{}])[0]
state.is_mono = stream.get('channels', 2) == 1
state.src_sr = int(stream.get('sample_rate', 44100))
state.src_br = int(stream.get('bit_rate', 128_000))
state.total_s = float(pr.get('format', {}).get('duration', 300))
# 2. Full-file 9-window probe (also yields rms_by_position)
full_spectrum, rms_by_pos = _probe_full_file(path, state.total_s, n_windows=9)
state.full_spectrum = full_spectrum
# 3. Adaptive window from rms_by_position
state.skip_s, state.dur_s = _adaptive_window(path, state.total_s, rms_by_pos)
# 4. Load clip (fast-seek, consistent for all clip measurements)
clip = load_audio_fast(path, skip_s=state.skip_s, duration_s=state.dur_s)
if len(clip) < SR * 3:
L(' [analyze] ⚠ clip too short — using defaults')
return state
# 5. Silence measurement
sil = _measure_silence(clip, state.total_s, state.skip_s)
state.silence_valid = sil['valid']
state.silence_floor = sil['floor']
state.silence_sfm = sil['sfm']
state.hum_freq_hz = sil['hum']
state.noise_type = sil['noise_type']
state.silence_frame_abs = sil.get('frame_positions', [])
# 6. Clip metrics
state.clip_rms = rms_db(clip)
state.clip_crest = crest_factor(clip)
state.clip_lra = lra_estimate(clip)
state.clip_sfm = compute_sfm(clip)
state.clip_dr = compute_dynamic_range(clip)
state.band_snr = compute_band_snr(clip)
state.snr_global = float(np.mean(list(state.band_snr.values()))) if state.band_snr else 25.0
# 7. Spectral characteristics from full_spectrum
spec = state.full_spectrum or third_octave(clip)
state.hf_rolloff = max(detect_hf_rolloff(spec, 12.0), 2000.0)
state.codec_cutoff = float(max(detect_hf_rolloff(spec, 6.0), 4000.0))
hf_bands = [fc for fc in spec if fc >= 8000]
if hf_bands and ref.third_oct:
hf_out = float(np.mean([spec.get(fc, -80) for fc in hf_bands]))
hf_ref = float(np.mean([ref.third_oct.get(fc, -60) for fc in hf_bands]))
state.hf_deficit = hf_ref - hf_out
else:
state.hf_deficit = 0.0
# 8. Clip ratio (clipping detection)
clipped_n = int(np.sum(np.abs(clip) > 0.99))
state.clip_ratio = float(clipped_n / max(len(clip), 1))
# 9. Smear detection
state.smear_score, state.smear_desc = _detect_smear(clip)
# 10. Spectral distance
if ref.third_oct:
common = [fc for fc in spec if fc in ref.third_oct and 80 <= fc <= min(12000, state.codec_cutoff * 0.9)]
if common:
out_arr = np.array([spec[fc] for fc in common])
ref_arr = np.array([ref.third_oct[fc] for fc in common])
loff = float(np.mean(ref_arr - out_arr))
aw = np.array([max(0.2, 1 + A_WEIGHT.get(fc, 0) / 10) for fc in common])
state.spec_dist = float(np.sum(aw * np.abs((ref_arr - out_arr) - loff)) / np.sum(aw))
# 11. Source tier (after all measurements)
state.source_tier = _derive_source_tier(
state.src_br, state.codec_cutoff, state.snr_global,
state.noise_type, state.smear_score)
# 12. Achievable targets
state.achievable_lufs, state.achievable_crest, state.achievable_lra = \
_compute_achievable(state.source_tier, state.codec_cutoff)
# 13. MDS (once, with tier-appropriate weights)
w = {'TIER_PRISTINE': {'snr': 0.25, 'sfm': 0.25, 'spec': 0.20, 'hf': 0.15, 'dr': 0.15},
'TIER_COMPRESSED': {'snr': 0.30, 'sfm': 0.25, 'spec': 0.18, 'hf': 0.07, 'dr': 0.20},
'TIER_DEGRADED': {'snr': 0.35, 'sfm': 0.28, 'spec': 0.15, 'hf': 0.02, 'dr': 0.20},
'TIER_DAMAGED': {'snr': 0.40, 'sfm': 0.30, 'spec': 0.10, 'hf': 0.00, 'dr': 0.20},
}.get(state.source_tier, {'snr': 0.25, 'sfm': 0.25, 'spec': 0.20, 'hf': 0.15, 'dr': 0.15})
sfm_ratio = state.clip_sfm / (ref.sfm + 1e-6)
mds = (float(np.clip((30.0 - state.snr_global) / 30.0, 0, 1)) * 100 * w['snr'] +
float(np.clip((sfm_ratio - 1.0) / 5.0, 0, 1)) * 100 * w['sfm'] +
float(np.clip(state.spec_dist / 15.0, 0, 1)) * 100 * w['spec'] +
float(np.clip(state.hf_deficit / 30.0, 0, 1)) * 100 * w['hf'] +
float(np.clip(max(0, state.clip_dr - ref.dr) / 8.0, 0, 1)) * 100 * w['dr'])
state.mds_raw = float(np.clip(mds, 0, 100))
# 14. Confidence vectors
_compute_confidence_vectors(state, ref)
return state
# ══════════════════════════════════════════════════════════════════════════════
# NR PASS (Phase B — dedicated, separate from EQ)
# ══════════════════════════════════════════════════════════════════════════════
def _build_hum_notch(sil: Dict) -> str:
"""Build hum notch filter from silence measurement data."""
if not sil.get('valid'): return ''
parts = []
if sil.get('hum_50db', 0) > 15.0:
parts.append(f'equalizer=f=50:width_type=q:width=0.4:g=-{min(24, int(sil["hum_50db"] * 0.8))}')
if sil.get('hum_60db', 0) > 15.0:
parts.append(f'equalizer=f=60:width_type=q:width=0.4:g=-{min(24, int(sil["hum_60db"] * 0.8))}')
if sil.get('hum_100db', 0) > 12.0:
parts.append(f'equalizer=f=100:width_type=q:width=0.6:g=-{min(18, int(sil["hum_100db"] * 0.7))}')
if sil.get('hum_120db', 0) > 12.0:
parts.append(f'equalizer=f=120:width_type=q:width=0.6:g=-{min(18, int(sil["hum_120db"] * 0.7))}')
return ','.join(parts)
def nr_pass(input_path: str, state: InputState, ref: ReferenceModel,
silence_data: Dict) -> Tuple[str, Dict]:
"""
Phase B: Dedicated NR pass — separate from EQ, always before EQ.
Returns (output_wav_path, nr_report).
"""
nr_report = {'applied': False, 'floor_delta': 0.0, 'sib_delta': 0.0, 'reverted': False}
if state.nr_confidence <= 0.05:
return input_path, nr_report
# Build NR filter string
ref_nr_floor = float(ref.silence_floor - 3.0) # never push below Sheikh's own floor
nf = float(np.clip(max(state.silence_floor + 2.0, ref_nr_floor), -76, -40))
max_nr = {'TIER_DAMAGED': 15, 'TIER_DEGRADED': 10, 'TIER_COMPRESSED': 6}.get(state.source_tier, 5)
nr_depth = max(3, min(max_nr, int(state.nr_confidence * max_nr)))
nr_filter = f'afftdn=nr={nr_depth}:nf={nf:.0f}:tn=1'
hum_notch = _build_hum_notch(silence_data)
# Combine hum notch + NR
filters = []
if hum_notch:
filters.append(hum_notch)
filters.append(nr_filter)
# Low-pass if severe codec ringing
if state.src_br < 65000 and state.hf_rolloff < 16000:
lp_hz = int(min(state.hf_rolloff * 0.97, 15000))
filters.append(f'lowpass=f={lp_hz}:poles=2')
full_filter = ','.join(filters)
tmp_nr = os.path.join(_TMP, 'v90_nr.wav')
ok = ffmpeg_process(input_path, tmp_nr, full_filter)
if not ok:
L(' [NR] ⚠ ffmpeg failed — bypass')
return input_path, nr_report
# Validate NR effectiveness using stored silence frame positions
# Measure pre-NR sibilant SNR
pre_clip = load_audio_fast(input_path, state.skip_s, min(30, state.dur_s))
pre_sib = compute_sibilant_snr(pre_clip, state.silence_floor)
# Load silence frames from NR output using absolute positions
post_floor_samples = []
for pos in state.silence_frame_abs[:10]: # sample first 10 silence positions
seg = load_audio_fast(tmp_nr, skip_s=pos, duration_s=0.2)
if len(seg) > 100:
post_floor_samples.append(rms_db(seg))
post_floor = float(np.median(post_floor_samples)) if post_floor_samples else state.silence_floor
post_clip = load_audio_fast(tmp_nr, state.skip_s, min(30, state.dur_s))
post_sib = compute_sibilant_snr(post_clip, post_floor)
floor_delta = state.silence_floor - post_floor # positive = floor dropped (good)
sib_delta = post_sib - pre_sib # negative = sibilant harmed (bad)
L(f' [NR] depth={nr_depth} nf={nf:.0f}dB floor: {state.silence_floor:.1f}{post_floor:.1f}'
f' (Δ={floor_delta:+.1f}dB) sib_snr: {pre_sib:.1f}{post_sib:.1f} (Δ={sib_delta:+.1f})')
# Do-no-harm gates
if sib_delta < -3.0:
L(' [NR] ⚠ sibilant drop > 3dB — REVERTED')
try: os.unlink(tmp_nr)
except: pass
nr_report['reverted'] = True
return input_path, nr_report
if rms_db(post_clip) - rms_db(pre_clip) > 1.0:
L(' [NR] ⚠ voiced RMS changed > 1dB — REVERTED')
try: os.unlink(tmp_nr)
except: pass
nr_report['reverted'] = True
return input_path, nr_report
if floor_delta < 2.0:
L(' [NR] ℹ floor delta < 2dB — NR had minimal effect')
nr_report.update({'applied': True, 'floor_delta': float(floor_delta),
'sib_delta': float(sib_delta)})
return tmp_nr, nr_report
# ══════════════════════════════════════════════════════════════════════════════
# EQ SYSTEM (Phase C — always post-NR)
# ══════════════════════════════════════════════════════════════════════════════
def _bias_band_weight(fc: float, codec_cutoff: float, hf_rolloff: float) -> float:
"""Per-band bias weight (0.0–1.0). 0.20 minimum floor below cutoff."""
if fc > hf_rolloff * 0.9: return 0.0
if fc > codec_cutoff: return 0.0
cutoff_safe = codec_cutoff * 0.85
if fc < cutoff_safe: return 1.0
raw = (codec_cutoff - fc) / max(codec_cutoff * 0.15, 1)
return max(0.20, float(np.clip(raw, 0, 1)))
def build_bias_filter_nodes(state: InputState) -> List[Tuple[float, float, float]]:
"""SPECTRAL_BIAS_V9 as parametric EQ nodes with per-band codec weighting."""
nodes = []
for fc, bias_db in SPECTRAL_BIAS_V9.items():
g = round(-bias_db * BIAS_SCALE, 2)
if abs(g) < 0.20: continue
w = _bias_band_weight(float(fc), state.codec_cutoff, state.hf_rolloff)
if w <= 0.0: continue
g_scaled = round(g * w, 2)
if abs(g_scaled) < 0.15: continue
Q = 0.65 if abs(g_scaled) > 1.5 else 0.90
nodes.append((float(fc), g_scaled, Q))
return nodes
def optimize_eq(inp_b: Dict, ref_b: Dict, n_nodes: int = 12, max_db: float = 6.0,
sib_cap: float = None, hf_ceil: float = 12000.0,
warmstart: List[Tuple] = None) -> List[Tuple]:
"""
scipy L-BFGS-B perceptual EQ optimizer (unchanged algorithm from v7.6).
v9.0 additions: warmstart from previous iteration, post-NR spectrum as input.
"""
if not SCIPY_OK: return []
ceil = min(hf_ceil, 12000.0)
common = sorted(fc for fc in inp_b if fc in ref_b and 63 <= fc <= ceil)
if len(common) < 4: return []
fc_arr = np.array(common, dtype=float)
inp_arr = np.array([inp_b[fc] for fc in common])
ref_arr = np.array([ref_b[fc] for fc in common])
loff = float(np.mean(ref_arr - inp_arr))
target = (ref_arr - inp_arr) - loff
def baw(fc):
bw = (2.0 if 500 <= fc <= 4000 else 1.6 if 200 <= fc < 500
else 1.4 if 4000 < fc <= 8000 else 0.9)
return bw * max(0.3, 1 + A_WEIGHT.get(fc, 0) / 10)
aw = np.array([baw(fc) for fc in common])
init_f = np.logspace(np.log10(63), np.log10(ceil), n_nodes)
def resp(fa, p):
r = np.zeros(len(fa))
for i in range(n_nodes):
f0 = abs(p[i * 3]) + 1e-6; g = p[i * 3 + 1]; Q = max(0.3, abs(p[i * 3 + 2]))
rat = fa / f0
r += g / (1 + Q ** 2 * (rat - 1.0 / (rat + 1e-9)) ** 2)
return r
def obj(p):
e = np.mean(aw * (resp(fc_arr, p) - target) ** 2)
gs = [p[i * 3 + 1] for i in range(n_nodes)]
sm = sum(0.012 * (gs[i + 1] - gs[i]) ** 2 for i in range(len(gs) - 1))
mg = sum(0.002 * g ** 2 for g in gs)
return e + sm + mg
# Warm start from previous iteration
if warmstart and len(warmstart) == n_nodes:
x0 = []
for f0, g, Q in warmstart:
x0.extend([float(np.clip(f0, 63, ceil)), float(np.clip(g, -max_db, max_db)), float(Q)])
else:
ig = np.interp(np.log10(init_f), np.log10(fc_arr), target)
x0 = []
for f, g in zip(init_f, ig):
x0.extend([float(np.clip(f, 63, ceil)), float(np.clip(g, -max_db, max_db)), 1.0])
res = minimize(obj, x0, method='L-BFGS-B',
bounds=[(63, ceil), (-max_db, max_db), (0.3, 4.5)] * n_nodes,
options={'maxiter': 500, 'ftol': 1e-10, 'gtol': 1e-9})
nodes = []
for i in range(n_nodes):
f0 = abs(res.x[i * 3]); g = res.x[i * 3 + 1]; Q = max(0.3, abs(res.x[i * 3 + 2]))
if sib_cap is not None and 2000 <= f0 <= 6300:
g = float(np.clip(g, -sib_cap, sib_cap))
if abs(g) >= 0.35:
nodes.append((round(f0, 0), round(g, 2), round(Q, 2)))
return sorted(nodes, key=lambda x: x[0])
def design_eq(post_nr_spectrum: Dict, ref: ReferenceModel, state: InputState,
warmstart: List[Tuple] = None) -> List[Tuple]:
"""
Phase C: EQ design from post-NR spectrum.
Bias nodes absorbed into the optimizer target (not applied separately).
Warmth correction included naturally via spectral target.
"""
# Build biased target: ref + bias correction
bias_nodes = build_bias_filter_nodes(state)
biased_target = dict(ref.third_oct)
for fc, g, _ in bias_nodes:
if fc in biased_target:
biased_target[fc] = biased_target[fc] + g # bias shifts target down/up
# Optimizer
sib_cap = 2.0 if state.smear_score >= 4.0 else None
hf_ceil = min(state.hf_rolloff * 0.9, ref.ref_codec_cutoff, 12000.0)
eq_nodes = optimize_eq(
post_nr_spectrum, biased_target,
n_nodes=12, max_db=6.0, sib_cap=sib_cap,
hf_ceil=hf_ceil, warmstart=warmstart)
# Scale all nodes by eq_confidence
eq_nodes = [(f, round(g * state.eq_confidence, 2), q) for f, g, q in eq_nodes]
eq_nodes = [(f, g, q) for f, g, q in eq_nodes if abs(g) >= 0.15]
return eq_nodes
def nodes_to_af(nodes: List[Tuple]) -> str:
"""Convert EQ node list to ffmpeg -af equalizer string."""
parts = []
for f0, g, Q in nodes:
if abs(g) < 0.10: continue
parts.append(f'equalizer=f={f0:.0f}:width_type=q:width={Q:.2f}:g={g:.2f}')
return ','.join(parts)
# ══════════════════════════════════════════════════════════════════════════════
# JOINT LUFS+LRA OPTIMIZER (Phase D)
# ══════════════════════════════════════════════════════════════════════════════
def _sample_compand_effect(eq_wav: str, curve_str: str, positions: List[float],
sample_s: float = 25.0) -> Tuple[float, float, float]:
"""Measure mean (lufs_delta, lra_delta, crest_delta) across positions."""
pre_lra = []; pre_lufs = []; post_lufs_l = []; post_lra_l = []; post_crest_l = []
for pos in positions:
pre_clip = load_audio_fast(eq_wav, skip_s=pos, duration_s=sample_s)
if len(pre_clip) < SR * 5: continue
pre_lra.append(lra_estimate(pre_clip))
pre_lufs.append(rms_db(pre_clip)) # use RMS as proxy for per-clip level
if not pre_lra: return 0.0, 0.0, 0.0
tmp = os.path.join(_TMP, f'v90_joint_{abs(hash(curve_str)) % 9999:04d}.wav')
af = f'compand=points={curve_str}'
ok = ffmpeg_process(eq_wav, tmp, af)
if not ok: return 0.0, 0.0, 0.0
for pos in positions:
c = load_audio_fast(tmp, skip_s=pos, duration_s=sample_s)
if len(c) < SR * 5: continue
post_lra_l.append(lra_estimate(c))
post_lufs_l.append(rms_db(c))
post_crest_l.append(crest_factor(c))
try: os.unlink(tmp)
except: pass
if not post_lra_l: return 0.0, 0.0, 0.0
lufs_delta = float(np.mean(post_lufs_l)) - float(np.mean(pre_lufs))
lra_delta = float(np.mean(post_lra_l)) - float(np.mean(pre_lra))
mean_crest = float(np.mean(post_crest_l))
return lufs_delta, lra_delta, mean_crest
def joint_lufs_lra_optimize(result_1: PassResult, ref: ReferenceModel,
state: InputState,
cached: JointParams = None) -> JointParams:
"""
3-position × 3-curve empirical PCHIP spline joint optimizer.
Returns JointParams with optimal compand_str + gain_db.
"""
# Re-use cache if LRA hasn't shifted significantly
if cached is not None:
if abs(result_1.lra - ref.phrase_lra_p50) < 0.3:
return cached
# Sample positions: skip_s, 40%, 70% through file
total = state.total_s
positions = [
float(state.skip_s),
float(total * 0.40),
float(total * 0.70),
]
positions = [min(p, total - 35) for p in positions]
if state.compand_confidence < 0.05:
# Compand bypass — only gain trim
gain_db = state.achievable_lufs - result_1.lufs
return JointParams(compand_str=_COMPAND_LIBRARY['BYPASS'],
gain_db=float(np.clip(gain_db, -6, 6)),
intensity_label='BYPASS')
# Sample LIGHT, MEDIUM, HEAVY
sample_curves = ['LIGHT', 'MEDIUM', 'HEAVY']
lra_deltas = []; lufs_deltas = []; crest_vals = []
for name in sample_curves:
ld, lrad, crt = _sample_compand_effect(
result_1.wav_path, _COMPAND_LIBRARY[name], positions)
lra_deltas.append(lrad); lufs_deltas.append(ld); crest_vals.append(crt)
L(f' [joint] {name}: LRA_Δ={lrad:+.2f} LUFS_Δ={ld:+.2f} Crest={crt:.2f}')
intensities = [_COMPAND_INTENSITY[n] for n in sample_curves]
# Target LRA delta
target_lra_delta = ref.phrase_lra_p50 - result_1.lra
L(f' [joint] target_LRA_Δ={target_lra_delta:+.2f} '
f'(current={result_1.lra:.2f}{ref.phrase_lra_p50:.2f})')
# Interpolate using PCHIP if available, else linear
if _PCHIP_OK and len(set(lra_deltas)) >= 2:
interp = PchipInterpolator(intensities, lra_deltas)
# Binary search for target intensity
lo, hi = 0.0, 1.0
for _ in range(30):
mid = (lo + hi) / 2
if float(interp(mid)) < target_lra_delta:
lo = mid
else:
hi = mid
target_intensity = (lo + hi) / 2
else:
# Linear fallback
target_intensity = float(np.interp(target_lra_delta, lra_deltas, intensities))
target_intensity = float(np.clip(target_intensity * state.compand_confidence, 0, 1))
# Select nearest named curve
best_name = min(_COMPAND_INTENSITY.keys(),
key=lambda n: abs(_COMPAND_INTENSITY[n] - target_intensity))
best_idx = sample_curves.index(best_name) if best_name in sample_curves else 1
# Predicted metrics at selected curve
predicted_lra_delta = float(np.interp(target_intensity, intensities, lra_deltas))
predicted_lufs_delta = float(np.interp(target_intensity, intensities, lufs_deltas))
predicted_crest = float(np.interp(target_intensity, intensities, crest_vals))
# Crest guard — if crest below achievable, reduce intensity
crest_guard_hit = False
if predicted_crest < state.achievable_crest - 0.5:
L(f' [joint] ⚠ crest guard: {predicted_crest:.2f} < {state.achievable_crest - 0.5:.2f}')
crest_guard_hit = True
# Step down intensity
lighter = {'EXTREME': 'HEAVY', 'HEAVY': 'MEDIUM', 'MEDIUM': 'LIGHT',
'LIGHT': 'MINIMAL', 'MINIMAL': 'BYPASS'}.get(best_name, 'BYPASS')
best_name = lighter
# Required gain to hit LUFS target
predicted_lufs = result_1.lufs + predicted_lufs_delta
gain_db = state.achievable_lufs - predicted_lufs
gain_db = float(np.clip(gain_db, -6.0, 6.0))
L(f' [joint] selected={best_name} intensity={target_intensity:.2f} '
f'gain={gain_db:+.2f}dB crest_guard={crest_guard_hit}')
return JointParams(
compand_str=_COMPAND_LIBRARY[best_name],
gain_db=gain_db,
predicted_lufs=predicted_lufs + gain_db,
predicted_lra=result_1.lra + predicted_lra_delta,
predicted_crest=predicted_crest,
intensity_label=best_name,
crest_guard_hit=crest_guard_hit,
)
# ══════════════════════════════════════════════════════════════════════════════
# PASS EXECUTION
# ══════════════════════════════════════════════════════════════════════════════
def run_pass_eq(nr_wav: str, eq_nodes: List[Tuple], pass_label: str = 'eq') -> str:
"""Apply EQ nodes to WAV. Returns output WAV path."""
af = nodes_to_af(eq_nodes)
if not af:
af = 'volume=1.0' # identity
# Soft limiter for WAV intermediate (not hard like final encode)
af += ',alimiter=limit=0.9997:level=false:attack=5:release=50'
out = os.path.join(_TMP, f'v90_{pass_label}.wav')
ok = ffmpeg_process(nr_wav, out, af)
if not ok:
L(f' [run_pass_eq] ⚠ failed — returning input')
return nr_wav
return out
def run_pass_joint(eq_wav: str, jp: JointParams, pass_label: str = 'joint') -> str:
"""Apply compand + gain. Returns output WAV path."""
parts = []
if jp.intensity_label != 'BYPASS':
parts.append(f'compand=points={jp.compand_str}')
if abs(jp.gain_db) > 0.05:
parts.append(f'volume={jp.gain_db:.3f}dB')
parts.append('alimiter=limit=0.9997:level=false:attack=5:release=50')
af = ','.join(parts)
out = os.path.join(_TMP, f'v90_{pass_label}.wav')
ok = ffmpeg_process(eq_wav, out, af)
if not ok:
L(f' [run_pass_joint] ⚠ failed — returning input')
return eq_wav
return out
def _find_peak_position(wav_path: str, total_s: float) -> float:
"""Find approximate position of maximum RMS energy for True Peak sampling."""
positions = [total_s * f for f in [0.20, 0.35, 0.50, 0.65, 0.80]]
best_pos, best_rms = positions[2], -99.0
for pos in positions:
if pos + 12 >= total_s: continue
seg = load_audio_fast(wav_path, skip_s=pos, duration_s=10)
r = rms_db(seg)
if r > best_rms:
best_rms = r; best_pos = pos
return best_pos
def run_pass_encode(best_wav: str, output_path: str,
state: InputState, ref: ReferenceModel) -> Tuple[str, float, int]:
"""
Final encode: alimiter + 320k MP3 with True Peak guarantee.
Returns (output_path, true_peak_db, n_retries).
Phase 4.9: adjust limiter threshold (not gain) on TP excess.
"""
# Final LUFS trim
measured_lufs = measure_lufs(best_wav)
lufs_trim = state.achievable_lufs - measured_lufs
lufs_trim = float(np.clip(lufs_trim, -6.0, 6.0))
limiter_threshold = 0.891
true_peak_db = -2.0
n_retries = 0
for attempt in range(3):
parts = []
if abs(lufs_trim) > 0.05:
parts.append(f'volume={lufs_trim:.3f}dB')
parts.append(f'alimiter=limit={limiter_threshold:.4f}:level=false:attack=1:release=15')
af = ','.join(parts)
sp, tc = _safe(best_wav)
cmd = ['ffmpeg', '-y', '-i', sp, '-af', af,
'-b:a', '320k', '-ar', '48000', '-ac', '2',
'-loglevel', 'error', output_path]
r = subprocess.run(cmd, capture_output=True)
if tc:
try: os.remove(tc)
except: pass
if r.returncode != 0 or not os.path.exists(output_path):
L(f' [encode] ⚠ attempt {attempt+1} failed')
continue
# Measure True Peak at loudest position
peak_pos = _find_peak_position(output_path, state.total_s)
sample = load_audio_fast(output_path, skip_s=peak_pos, duration_s=30)
true_peak_db = float(20 * np.log10(np.max(np.abs(sample)) + 1e-10))
if true_peak_db <= -0.5:
break
else:
n_retries += 1
# Exact limiter threshold correction (Phase 4.9)
excess_db = true_peak_db - (-1.0)
limiter_threshold = limiter_threshold * (10 ** (-excess_db / 20))
limiter_threshold = max(0.700, limiter_threshold)
L(f' [encode] TP={true_peak_db:.2f}dBTP > -0.5 → limiter→{limiter_threshold:.4f}')
return output_path, true_peak_db, n_retries
# ══════════════════════════════════════════════════════════════════════════════
# PASS MEASUREMENT
# ══════════════════════════════════════════════════════════════════════════════
def measure_pass(wav_path: str, ref: ReferenceModel, state: InputState,
pass_label: str = '', is_final: bool = False) -> PassResult:
"""
Full PassResult from WAV or MP3.
Final pass: 9-window spectrum. Intermediates: 3-window.
Clip measurements always from consistent (skip_s, dur_s) window.
"""
result = PassResult(pass_label=pass_label, wav_path=wav_path)
# Spectrum
if is_final:
spectrum, _ = _probe_full_file(wav_path, state.total_s, n_windows=9)
else:
spectrum = _probe_3window(wav_path, state.total_s, state.skip_s)
if not spectrum:
spectrum = {}
result.spectrum = spectrum
# Clip measurements
clip = load_audio_fast(wav_path, skip_s=state.skip_s, duration_s=state.dur_s)
if len(clip) < SR * 3:
return result
result.rms = float(rms_db(clip))
result.crest = float(crest_factor(clip))
result.lra = float(lra_estimate(clip))
result.lufs = float(measure_lufs(wav_path))
# EQ residual
ref_b = ref.third_oct
common = [fc for fc in spectrum if fc in ref_b and 80 <= fc <= min(12000, state.codec_cutoff * 0.9)]
if common:
out_arr = np.array([spectrum[fc] for fc in common])
ref_arr = np.array([ref_b[fc] for fc in common])
loff = float(np.mean(ref_arr - out_arr))
result.eq_residual = float(np.mean(np.abs((ref_arr - out_arr) - loff)))
else:
result.eq_residual = 20.0
# Sibilant SNR
result.sib_snr = float(compute_sibilant_snr(clip, state.silence_floor))
# Scores
result.score_tier, result.score_abs, result.ceiling_reason = _quality_score_v90(
spectrum, result.lufs, result.rms, result.crest, result.lra, ref, state)
# Composite (tier-weighted)
tier_weights = {
'TIER_PRISTINE': (0.45, 0.35, 0.20),
'TIER_COMPRESSED': (0.35, 0.40, 0.25),
'TIER_DEGRADED': (0.25, 0.40, 0.35),
'TIER_DAMAGED': (0.15, 0.45, 0.40),
}
wc, we, wl = tier_weights.get(state.source_tier, (0.35, 0.40, 0.25))
crest_norm = float(np.clip((result.crest - 5.0) / max(state.achievable_crest - 5.0, 0.1), 0, 1.2))
eq_norm = max(0.0, 1.0 - result.eq_residual / 5.0)
lra_norm = max(0.0, 1.0 - abs(result.lra - ref.phrase_lra_p50) / 3.0)
result.composite = crest_norm * wc + eq_norm * we + lra_norm * wl
return result
# ══════════════════════════════════════════════════════════════════════════════
# CONVERGENCE CONTROL
# ══════════════════════════════════════════════════════════════════════════════
def should_stop(history: List[PassResult], state: InputState,
ref: ReferenceModel) -> Tuple[bool, str]:
n = len(history)
if n >= 6: return True, 'max_passes'
if n < 1: return False, ''
last = history[-1]
if last.crest < state.achievable_crest - 1.5:
return True, 'crest_collapsed'
if n >= 2 and last.eq_residual > history[-2].eq_residual + 0.15:
return True, 'oscillation'
if n >= 3:
if (history[-1].composite < history[-2].composite - 0.02
and history[-2].composite < history[-3].composite - 0.02):
return True, 'composite_regression'
# Joint convergence (Phase 3.9: EQ alone not enough)
lufs_ok = abs(last.lufs - state.achievable_lufs) < 0.30
lra_ok = abs(last.lra - ref.phrase_lra_p50) < 0.30
eq_ok = last.eq_residual < 0.40
if lufs_ok and lra_ok and eq_ok:
return True, 'fully_converged'
if eq_ok and lra_ok:
return True, 'converged_eq_lra'
if n >= 2 and last.composite > history[-2].composite + 0.01:
return False, '' # still improving
return True, 'default_stop'
# ══════════════════════════════════════════════════════════════════════════════
# SCORING
# ══════════════════════════════════════════════════════════════════════════════
def _quality_score_v90(spectrum: Dict, lufs: float, rms: float,
crest: float, lra: float,
ref: ReferenceModel, state: InputState) -> Tuple[float, float, str]:
"""5-component quality score vs tier-achievable and absolute targets."""
ref_b = ref.third_oct
# 1. Spectral (30 pts)
common = [fc for fc in spectrum if fc in ref_b
and 80 <= fc <= min(12000, state.codec_cutoff * 0.9)]
if common:
out_arr = np.array([spectrum[fc] for fc in common])
ref_arr = np.array([ref_b[fc] for fc in common])
aw = np.array([max(0.2, 1 + A_WEIGHT.get(fc, 0) / 10) for fc in common])
loff = float(np.mean(ref_arr - out_arr))
avg_err = float(np.sum(aw * np.abs((ref_arr - out_arr) - loff)) / np.sum(aw))
else:
avg_err = 99.0
spectral_score = 30.0 * max(0.0, 1.0 - avg_err / 6.0)
# 2–4. LUFS / Crest / LRA — vs achievable targets (tier-honest)
lufs_score = 25.0 * max(0.0, 1.0 - abs(lufs - state.achievable_lufs) / 3.0)
crest_score = 20.0 * max(0.0, 1.0 - abs(crest - state.achievable_crest) / 3.0)
lra_score = 15.0 * max(0.0, 1.0 - abs(lra - ref.phrase_lra_p50) / 2.5)
# 5. Warmth tilt (10 pts)
tfc = np.array([fc for fc in CENTERS_31 if 200 <= fc <= 2000 and fc in spectrum], dtype=float)
if len(tfc) >= 3:
inp_tilt = float(np.polyfit(np.log2(tfc / 1000.0),
np.array([spectrum[fc] for fc in tfc]), 1)[0])
warmth_score = 10.0 * max(0.0, 1.0 - abs(inp_tilt - ref.warmth_ratio) / 3.0)
else:
warmth_score = 5.0
score_tier = round(spectral_score + lufs_score + crest_score + lra_score + warmth_score, 1)
# Absolute score (vs 1425H targets, not tier-adjusted)
lufs_abs = 25.0 * max(0.0, 1.0 - abs(lufs - TARGET['lufs']) / 3.0)
crest_abs = 20.0 * max(0.0, 1.0 - abs(crest - TARGET['crest']) / 3.0)
lra_abs = 15.0 * max(0.0, 1.0 - abs(lra - ref.phrase_lra_p50) / 2.5)
score_abs = round(spectral_score + lufs_abs + crest_abs + lra_abs + warmth_score, 1)
ceiling_reason = ''
if state.source_tier != 'TIER_PRISTINE' and score_tier > score_abs + 2.0:
ceiling_reason = (f'{state.source_tier}: Crest≤{state.achievable_crest:.2f} '
f'LRA≤{state.achievable_lra:.2f} LUFS≥{state.achievable_lufs:.2f}')
return score_tier, score_abs, ceiling_reason
# ══════════════════════════════════════════════════════════════════════════════
# MAIN ENTRY POINT: enhance()
# ══════════════════════════════════════════════════════════════════════════════
def enhance(input_path: str, output_path: str,
max_iterations: int = 3, target_score: float = 96.0) -> Dict:
t0 = time.time()
MAX_T = 1800 # 30 minute hard timeout
def _chk(phase):
if time.time() - t0 > MAX_T:
raise TimeoutError(f'enhance() exceeded {MAX_T}s at phase={phase}')
L(f'╔══════════════════════════════════════════════════╗')
L(f'║ Audio Enhancement Engine v9.0 — "التطور" ║')
L(f'║ المرجع: الشيخ ياسر الدوسري — 1425H ║')
L(f'╚══════════════════════════════════════════════════╝')
L(f' الملف: {os.path.basename(input_path)}')
# ── PHASE A: Reference + Input Analysis ────────────────────────────────
L('\nPass 1 — تحليل المدخل والمرجع')
_chk('phase_A')
ref = load_reference_model()
L(f' ✓ مرجع: {ref.n_files} ملف | RMS={ref.rms:.2f} Crest={ref.crest:.2f} '
f'LRA={ref.lra:.2f} p50={ref.phrase_lra_p50:.2f}')
state = analyze_input(input_path, ref)
L(f' ✓ {state.total_s:.0f}s | {state.source_tier} | '
f'cutoff={state.codec_cutoff:.0f}Hz | smear={state.smear_score}/10 ({state.smear_desc})')
L(f' ✦ Crest={state.clip_crest:.2f} LRA={state.clip_lra:.2f} '
f'SNR={state.snr_global:.1f}dB noise={state.noise_type}')
L(f' ✦ eq={state.eq_confidence:.2f} nr={state.nr_confidence:.2f} '
f'compand={state.compand_confidence:.2f} hf={state.hf_confidence:.2f}')
L(f' ✦ achievable: LUFS≥{state.achievable_lufs:.2f} '
f'Crest≤{state.achievable_crest:.2f} LRA≤{state.achievable_lra:.2f}')
L(f' ✦ MDS={state.mds_raw:.1f}/100 spec_dist=±{state.spec_dist:.2f}dB')
# Silence data for NR (need raw dict, re-measure briefly)
clip = load_audio_fast(input_path, state.skip_s, state.dur_s)
silence_data = _measure_silence(clip, state.total_s, state.skip_s)
del clip
# ── PHASE B: NR Pass ────────────────────────────────────────────────────
L('\nPass 2 — تقليل الضوضاء (NR)')
_chk('phase_B')
nr_wav, nr_report = nr_pass(input_path, state, ref, silence_data)
if nr_report['applied']:
L(f' ✓ NR applied: floor_Δ={nr_report["floor_delta"]:+.1f}dB '
f'sib_Δ={nr_report["sib_delta"]:+.1f}dB')
else:
L(f' ✦ NR bypass (confidence={state.nr_confidence:.2f})')
# ── PHASE C: EQ Design (post-NR spectrum) ──────────────────────────────
L('\nPass 3 — تصميم التوازن الطيفي (post-NR)')
_chk('phase_C')
post_nr_spectrum, _ = _probe_full_file(nr_wav, state.total_s, n_windows=5)
if not post_nr_spectrum:
post_nr_spectrum = state.full_spectrum
eq_nodes = design_eq(post_nr_spectrum, ref, state)
L(f' ✓ EQ: {len(eq_nodes)} نود | eq_conf={state.eq_confidence:.2f}')
for f0, g, Q in sorted(eq_nodes, key=lambda x: x[0]):
L(f' {f0:.0f}Hz {g:+.2f}dB Q={Q:.2f}')
eq_warmstart = eq_nodes # warm start for subsequent iterations
# ── PHASE D: Iteration Loop ─────────────────────────────────────────────
L('\nPass 4 — التكرار التحسيني')
_chk('phase_D')
pass_history: List[PassResult] = []
best_wav = nr_wav
best_composite = -999.0
best_result: Optional[PassResult] = None
cached_joint: Optional[JointParams] = None
for iteration in range(max(1, max_iterations)):
_chk(f'iter_{iteration}')
L(f'\n ── Iteration {iteration + 1}/{max_iterations} ──')
# Pass D1: EQ application
eq_wav = run_pass_eq(nr_wav, eq_nodes, f'eq_{iteration}')
r1 = measure_pass(eq_wav, ref, state, f'EQ-{iteration+1}')
pass_history.append(r1)
L(f' [EQ] Crest={r1.crest:.2f} LRA={r1.lra:.2f} LUFS={r1.lufs:.2f} '
f'EQres={r1.eq_residual:.2f} comp={r1.composite:.3f}')
if r1.composite > best_composite:
best_composite = r1.composite
best_wav = eq_wav
best_result = r1
stop, reason = should_stop(pass_history, state, ref)
if stop and iteration == 0 and reason not in ('crest_collapsed',):
L(f' [stop?] early but continuing for joint pass...')
elif stop and iteration > 0:
L(f' [stop] {reason}')
break
# Pass D2: Joint LUFS+LRA
L(f' [joint] calibrating...')
joint_params = joint_lufs_lra_optimize(r1, ref, state, cached=cached_joint)
cached_joint = joint_params
joint_wav = run_pass_joint(eq_wav, joint_params, f'joint_{iteration}')
r2 = measure_pass(joint_wav, ref, state, f'Joint-{iteration+1}')
L(f' [Joint] Crest={r2.crest:.2f} LRA={r2.lra:.2f} LUFS={r2.lufs:.2f} '
f'EQres={r2.eq_residual:.2f} comp={r2.composite:.3f}')
# do-no-harm gate (Phase 4.8: compare to r1, not just to r2)
if r2.composite < r1.composite - 1.0:
L(f' [do-no-harm] joint degraded — trying half intensity')
half = JointParams(
compand_str=_COMPAND_LIBRARY.get(joint_params.intensity_label, joint_params.compand_str),
gain_db=joint_params.gain_db * 0.5,
intensity_label=joint_params.intensity_label)
half_wav = run_pass_joint(eq_wav, half, f'half_{iteration}')
r2h = measure_pass(half_wav, ref, state, f'Half-{iteration+1}')
if r2h.composite > r1.composite: # compare to r1 (Phase 4.8 fix)
joint_wav = half_wav; r2 = r2h
L(f' [do-no-harm] half-intensity accepted: comp={r2.composite:.3f}')
else:
joint_wav = eq_wav; r2 = r1 # full revert to Pass D1
L(f' [do-no-harm] reverted to EQ-only')
pass_history.append(r2)
if r2.composite > best_composite:
best_composite = r2.composite
best_wav = joint_wav
best_result = r2
stop, reason = should_stop(pass_history, state, ref)
if stop:
L(f' [stop] {reason}')
break
# Adaptive EQ refinement for next iteration
if iteration < max_iterations - 1:
if r2.eq_residual > 0.8 and r2.spectrum:
scale = min(0.35, 0.10 + (r2.eq_residual - 1.5) * 0.08)
# Warm-start from previous eq_nodes
eq_nodes_new = design_eq(r2.spectrum, ref, state,
warmstart=eq_nodes if len(eq_nodes) == 12 else None)
if eq_nodes_new:
eq_nodes = eq_nodes_new
eq_warmstart = eq_nodes
L(f' [refine] EQ refined: {len(eq_nodes)} nodes (scale={scale:.2f})')
if best_result is None:
best_result = pass_history[-1] if pass_history else PassResult()
# ── PHASE E: Final Encode ────────────────────────────────────────────────
L(f'\nPass 4 — الترميز النهائي MP3 320kbps')
_chk('phase_E')
output_path, true_peak_db, encode_retries = run_pass_encode(
best_wav, output_path, state, ref)
L(f' ✓ TP={true_peak_db:.2f}dBTP retries={encode_retries}')
# ── PHASE E: Final Score ─────────────────────────────────────────────────
final = measure_pass(output_path, ref, state, 'final', is_final=True)
elapsed = time.time() - t0
# Build summary
lines = [
f'v9.0 | {state.source_tier} | {elapsed:.0f}s',
f'Score: {final.score_tier:.0f}/100' +
(f' ({final.ceiling_reason})' if final.ceiling_reason else ''),
f'LUFS={final.lufs:.2f} Crest={final.crest:.2f} LRA={final.lra:.2f}',
]
if nr_report['applied']:
lines.append(f'NR: floor_Δ={nr_report["floor_delta"]:+.1f}dB '
f'smear={state.smear_score}/10 ({state.smear_desc})')
if encode_retries > 0:
lines.append(f'TP: {true_peak_db:.2f}dBTP | {encode_retries} retry(s)')
summary = '\n'.join(lines)
L(f'\n{"═"*50}')
L(f' LUFS={final.lufs:.2f} RMS={final.rms:.2f} Crest={final.crest:.2f} LRA={final.lra:.2f}')
L(f' ★ {final.score_tier:.1f}/100 ({state.source_tier})')
if final.ceiling_reason:
L(f' [ceiling] {final.ceiling_reason}')
L(f' [{elapsed:.1f}s | passes={len(pass_history)} | NR={nr_report["applied"]}]')
L(f'{"═"*50}')
return {
'engine_version': 'v9.0',
'score': final.score_tier,
'score_tier': final.score_tier,
'score_absolute': final.score_abs,
'ceiling_reason': final.ceiling_reason,
'lufs': final.lufs,
'rms': final.rms,
'crest': final.crest,
'lra': final.lra,
'true_peak_db': true_peak_db,
'encode_retries': encode_retries,
'source_tier': state.source_tier,
'eq_confidence': state.eq_confidence,
'nr_confidence': state.nr_confidence,
'compand_confidence': state.compand_confidence,
'smear_score': state.smear_score,
'smear_desc': state.smear_desc,
'codec_cutoff_hz': state.codec_cutoff,
'noise_type': state.noise_type,
'silence_floor_db': state.silence_floor,
'nr_applied': nr_report['applied'],
'nr_floor_delta_db': nr_report['floor_delta'],
'passes_used': len(pass_history),
'processing_time_s': round(elapsed, 1),
'eq_residual_final': final.eq_residual,
'mds': state.mds_raw,
'summary': summary,
}
# ══════════════════════════════════════════════════════════════════════════════
# COMPATIBILITY ALIAS (for any code that calls get_reference_fingerprint)
# ══════════════════════════════════════════════════════════════════════════════
def get_reference_fingerprint():
"""Compatibility alias — returns ReferenceModel as v8.x duck-typed object."""
return load_reference_model()
def _build_ref_cache_if_needed():
"""Called at Docker build time to pre-warm the cache."""
if not REF_FILES:
return
if os.path.exists(_REF_CACHE):
try:
with open(_REF_CACHE) as f:
d = json.load(f)
if d.get('ref_hash') == _ref_files_hash(REF_FILES):
return # already valid
except Exception:
pass
load_reference_model()
# ══════════════════════════════════════════════════════════════════════════════
# CLI ENTRY POINT
# ══════════════════════════════════════════════════════════════════════════════
def main() -> int:
if not NUMPY_OK or not SCIPY_OK:
print('pip install numpy scipy'); return 1
p = argparse.ArgumentParser(description='Audio Enhancement Engine v9.0 — 1425H')
p.add_argument('-i', '--input')
p.add_argument('-o', '--output')
p.add_argument('--iterations', type=int, default=3)
p.add_argument('--target', type=float, default=96.0)
p.add_argument('--ref', action='append', default=[], metavar='REF_MP3')
p.add_argument('--clear-cache', action='store_true')
args = p.parse_args()
if args.ref:
valid = [r for r in args.ref if os.path.exists(r)]
if valid:
global REF_FILES
REF_FILES = valid
if args.clear_cache:
if os.path.exists(_REF_CACHE):
os.remove(_REF_CACHE)
print('✅ Cache v9.0 deleted')
return 0
if not args.input or not args.output:
p.print_help(); return 1
try:
r = enhance(args.input, args.output, args.iterations, args.target)
print(f'\n ★ {r["score"]:.1f}/100 '
f'LUFS={r["lufs"]:.2f} RMS={r["rms"]:.2f} '
f'Crest={r["crest"]:.2f} LRA={r["lra"]:.2f}')
return 0 if r['score'] >= 85 else 1
except Exception as e:
print(f'❌ {e}'); return 1
if __name__ == '__main__':
sys.exit(main())