Open-Unmix / app.py
harp-dev's picture
Deploy HARP wrapper via model agent
1220b6a verified
Raw
History Blame Contribute Delete
3.89 kB
from __future__ import annotations
import gradio as gr
try:
import spaces
except ImportError: # 'spaces' is only provided by Hugging Face Spaces
import types as _types
def _gpu(*args, **kwargs):
if len(args) == 1 and callable(args[0]) and not kwargs:
return args[0]
def _decorator(func):
return func
return _decorator
spaces = _types.SimpleNamespace(GPU=_gpu)
from pyharp import *
try: # torch>=2.6 flipped torch.load(weights_only) to True; legacy ckpts need False
import torch as _torch
if getattr(_torch.load, "__harp_compat__", False) is False:
_torch_load_orig = _torch.load
def _torch_load_compat(*args, **kwargs):
kwargs.setdefault("weights_only", False)
return _torch_load_orig(*args, **kwargs)
_torch_load_compat.__harp_compat__ = True
_torch.load = _torch_load_compat
except Exception: # torch not installed / unexpected API -- nothing to patch
pass
import tempfile
import torch
import torchaudio
from openunmix import utils, predict
DEVICE = "cuda" if torch.cuda.is_available() else "cpu"
_separators = {}
def get_separator(model_name, niter, wiener_win_len):
key = (model_name, niter, wiener_win_len)
if key not in _separators:
sep = utils.load_separator(
model_str_or_path=model_name,
niter=niter,
wiener_win_len=wiener_win_len,
device=DEVICE,
pretrained=True,
)
sep.freeze()
sep.to(DEVICE)
_separators[key] = sep
return _separators[key]
model_card = ModelCard(
name="Open-Unmix",
description="Music source separation into vocals, drums, bass, and other instruments using deep neural networks.",
author="sigsep",
tags=["audio-to-audio", "source-separation", "stems"],
)
@spaces.GPU
def process_fn(input_audio, model_name, niter, wiener_win_len):
audio, rate = torchaudio.load(input_audio)
if audio.shape[0] == 1:
audio = audio.repeat(2, 1)
separator = get_separator(model_name, int(niter), int(wiener_win_len))
estimates = predict.separate(
audio=audio,
rate=rate,
model_str_or_path=model_name,
separator=separator,
device=DEVICE
)
output_files = {}
for target in ["vocals", "drums", "bass", "other"]:
target_tensor = estimates[target][0].cpu()
out_file = tempfile.NamedTemporaryFile(suffix=".wav", delete=False)
out_file.close()
torchaudio.save(out_file.name, target_tensor, int(separator.sample_rate))
output_files[target] = out_file.name
return (
output_files["vocals"],
output_files["drums"],
output_files["bass"],
output_files["other"]
)
with gr.Blocks() as demo:
input_components = [
gr.Audio(type="filepath", label="Input Audio").harp_required(True),
gr.Dropdown(choices=["umxl", "umxhq", "umx"], value="umxl", label="Model", info="umxl is the largest and best performing model, trained on extra data."),
gr.Slider(minimum=0, maximum=5, step=1, value=1, label="Wiener Filter Iterations", info="Number of iterations for Wiener filtering. 0 means softmask."),
gr.Slider(minimum=100, maximum=500, step=50, value=300, label="Wiener Window Length", info="Number of frames on which to apply filtering independently."),
]
output_components = [
gr.Audio(type="filepath", label="Vocals"),
gr.Audio(type="filepath", label="Drums"),
gr.Audio(type="filepath", label="Bass"),
gr.Audio(type="filepath", label="Other"),
]
build_endpoint(
model_card=model_card,
input_components=input_components,
output_components=output_components,
process_fn=process_fn,
)
demo.queue().launch(share=True, show_error=False, pwa=True)