import os import glob import gradio as gr import numpy as np import soundfile as sf import noisereduce as nr from audio_separator.separator import Separator # ----------------------------- # Locate the local ONNX model already committed to this repo # (avoids any network download and avoids hardcoding a filename # that may have changed in past rename commits) # ----------------------------- MODELS_DIR = "models" onnx_candidates = sorted(glob.glob(os.path.join(MODELS_DIR, "*.onnx"))) if not onnx_candidates: raise FileNotFoundError( f"No .onnx model found in '{MODELS_DIR}/'. " f"Make sure your UVR-MDX-Net model file is committed to that folder." ) MODEL_PATH = onnx_candidates[0] MODEL_DIR = os.path.dirname(MODEL_PATH) or "." MODEL_FILENAME = os.path.basename(MODEL_PATH) print(f"Using local model: {MODEL_PATH}") # ----------------------------- # Load the separator once at startup # (model_file_dir points at the folder that ALREADY has the file, # so audio-separator uses it directly instead of trying to download it) # ----------------------------- separator = Separator( output_dir="/tmp/audio_separator_outputs", model_file_dir=MODEL_DIR, ) separator.load_model(model_filename=MODEL_FILENAME) def denoise_vocals(vocals_path: str) -> str: """ Run spectral-gating noise reduction on the isolated vocals track. Uses `noisereduce` instead of a neural denoiser (e.g. DeepFilterNet2) because that pulled in a conflicting, unmaintained torch/torchaudio/onnx dependency chain that could no longer be resolved alongside audio-separator. noisereduce has no heavy ML dependencies, so it can't collide with audio-separator's stack the same way. """ audio, sr = sf.read(vocals_path) # soundfile returns shape (frames, channels) for stereo; noisereduce # expects channels-first for multi-channel input. is_stereo = audio.ndim == 2 y = audio.T if is_stereo else audio reduced = nr.reduce_noise(y=y, sr=sr) if is_stereo: reduced = reduced.T denoised_path = os.path.join( os.path.dirname(vocals_path), f"denoised_{os.path.basename(vocals_path)}", ) sf.write(denoised_path, reduced, sr) return denoised_path def separate_audio(audio_file): if audio_file is None: return None, None raw_outputs = separator.separate(audio_file) # separator.separate() returns bare filenames, not full paths — they # actually get written under separator.output_dir. Without joining them # back to that directory, Gradio resolves the relative filename against # the app's cwd instead of where the files really are, which is what # caused the empty players / 403 on download. output_paths = [ p if os.path.isabs(p) else os.path.join(separator.output_dir, p) for p in raw_outputs ] output_paths = [os.path.abspath(p) for p in output_paths] vocals_path = next((p for p in output_paths if "vocal" in p.lower()), None) instrumental_path = next( (p for p in output_paths if p != vocals_path), None ) if vocals_path is not None: try: vocals_path = denoise_vocals(vocals_path) except Exception as e: # If denoising fails for any reason, fall back to the raw # separated vocals rather than losing the whole result. print(f"Denoising failed, returning raw vocals: {e}") return vocals_path, instrumental_path # ----------------------------- # Gradio Interface # ----------------------------- ui = gr.Interface( fn=separate_audio, inputs=gr.Audio(type="filepath", label="Upload Audio"), outputs=[ gr.Audio(type="filepath", label="Vocals"), gr.Audio(type="filepath", label="Instrumental"), ], title="Vocal / Instrumental Separator", description=( f"CPU-friendly vocal & instrumental separation using the local " f"UVR-MDX-Net ONNX model ({MODEL_FILENAME}), with vocals cleaned " f"up via spectral-gating noise reduction." ), ) if __name__ == "__main__": # Explicitly whitelist the output directory for Gradio's file server — # avoids 403s when serving files from a directory outside the app's cwd. ui.launch(allowed_paths=[separator.output_dir])