GAME / app.py
LING
Deploy GAME singing-to-MIDI endpoint
fa73431
Raw
History Blame Contribute Delete
3.86 kB
from __future__ import annotations
import uuid
from pathlib import Path
import gradio as gr
import soundfile as sf
try:
import spaces
except ImportError:
class spaces:
class GPU:
def __init__(self, func=None, duration=60):
self.func = func
def __call__(self, *args, **kwargs):
if self.func is not None:
return self.func(*args, **kwargs)
return args[0]
from game_runtime import transcribe as transcribe_with_game
from pyharp import ModelCard, build_endpoint
MAX_AUDIO_SECONDS = 30
OUTPUT_ROOT = Path("/tmp/game_outputs")
LANGUAGE_CODES = {
"Automatic": "",
"English": "en",
"Japanese": "ja",
"Cantonese": "yue",
"Mandarin Chinese": "zh",
}
model_card = ModelCard(
name="GAME",
description=(
"Transcribe a monophonic singing voice recording "
"into editable MIDI and detailed note data."
),
author="OpenVPI",
tags=[
"music-information-retrieval",
"singing-voice",
"transcription",
"audio-to-midi",
],
)
@spaces.GPU(duration=120)
def process_fn(
audio_path: str | None,
language: str,
steps: str,
) -> tuple[str, str]:
if not audio_path:
raise gr.Error("Please upload a singing voice recording.")
try:
duration = sf.info(audio_path).duration
except Exception as exc:
raise gr.Error(f"Could not read the uploaded audio: {exc}") from exc
if duration > MAX_AUDIO_SECONDS:
raise gr.Error(
f"Audio must be no longer than {MAX_AUDIO_SECONDS} seconds. "
f"Received {duration:.1f} seconds."
)
output_dir = OUTPUT_ROOT / uuid.uuid4().hex
output_dir.mkdir(parents=True, exist_ok=True)
try:
transcribe_with_game(
audio_path=Path(audio_path),
output_dir=output_dir,
language_code=LANGUAGE_CODES[language],
steps=int(steps),
)
except Exception as exc:
raise gr.Error(f"GAME inference failed: {exc}") from exc
midi_files = sorted(output_dir.glob("*.mid"))
csv_files = sorted(output_dir.glob("*.csv"))
if not midi_files or not csv_files:
raise gr.Error("GAME did not produce the expected MIDI and CSV files.")
return str(midi_files[0]), str(csv_files[0])
with gr.Blocks(title="GAME Singing to MIDI") as demo:
input_components = [
gr.Audio(
type="filepath",
label="Singing Voice",
)
.harp_required(True)
.set_info(
"A monophonic singing voice recording, up to 30 seconds long."
),
gr.Dropdown(
choices=list(LANGUAGE_CODES),
value="Automatic",
label="Language",
info="Use Automatic when the singing language is unknown.",
),
gr.Dropdown(
choices=["1", "2", "4", "8"],
value="2",
label="D3PM Steps",
info=(
"More steps may improve transcription at the cost "
"of longer processing time."
),
),
]
output_components = [
gr.File(
type="filepath",
file_types=[".mid", ".midi"],
label="Transcribed MIDI",
).set_info("Editable MIDI transcription."),
gr.File(
type="filepath",
file_types=[".csv"],
label="Detailed Note Data",
).set_info("Note timing and pitch data generated by GAME."),
]
build_endpoint(
model_card=model_card,
input_components=input_components,
output_components=output_components,
process_fn=process_fn,
)
if __name__ == "__main__":
demo.queue(default_concurrency_limit=1).launch(
show_error=True,
pwa=True,
)