File size: 3,140 Bytes
33f35c7
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
from __future__ import annotations

import json
import tempfile
import uuid
from pathlib import Path

import gradio as gr
import soundfile as sf

try:
    import spaces
except ImportError:
    class spaces:
        class GPU:
            def __init__(self, func=None, duration=60):
                self.func = func

            def __call__(self, *args, **kwargs):
                if self.func is not None:
                    return self.func(*args, **kwargs)
                return args[0]

from pyharp import ModelCard, build_endpoint

from music2emo_runtime import analyze_music

MIN_AUDIO_SECONDS = 1
MAX_AUDIO_SECONDS = 60
OUTPUT_ROOT = Path(tempfile.gettempdir()) / "music2emo_outputs"

model_card = ModelCard(
    name="Music2Emo",
    description=(
        "Recognize music emotion as mood tags and continuous "
        "valence-arousal scores."
    ),
    author="AMAAI Lab",
    tags=[
        "music-information-retrieval",
        "music-emotion-recognition",
        "mood-tagging",
        "valence",
        "arousal",
    ],
)


def _validate_audio(path: str | None) -> str:
    if not path:
        raise gr.Error("Please upload an audio file.")
    try:
        duration = sf.info(path).duration
    except Exception as exc:
        raise gr.Error(f"Could not read the audio file: {exc}") from exc
    if duration < MIN_AUDIO_SECONDS:
        raise gr.Error("Audio must be at least 1 second long.")
    if duration > MAX_AUDIO_SECONDS:
        raise gr.Error(
            f"Audio must be no longer than {MAX_AUDIO_SECONDS} seconds. "
            f"Received {duration:.1f} seconds."
        )
    return path


@spaces.GPU(duration=120)
def process_fn(audio_path: str | None, threshold: float) -> str:
    audio_path = _validate_audio(audio_path)
    try:
        result = analyze_music(audio_path, threshold)
    except Exception as exc:
        raise gr.Error(f"Music2Emo inference failed: {exc}") from exc

    output_dir = OUTPUT_ROOT / uuid.uuid4().hex
    output_dir.mkdir(parents=True, exist_ok=True)
    output_path = output_dir / "music2emo_analysis.json"
    output_path.write_text(json.dumps(result, indent=2) + "\n", encoding="utf-8")
    return str(output_path)


with gr.Blocks(title="Music2Emo") as demo:
    input_components = [
        gr.Audio(type="filepath", label="Music Audio")
        .harp_required(True)
        .set_info("Music clip between 1 and 60 seconds."),
        gr.Slider(
            minimum=0.1,
            maximum=0.9,
            value=0.5,
            step=0.05,
            label="Mood Threshold",
        ).set_info("Minimum probability for a mood tag to be returned."),
    ]
    output_components = [
        gr.File(
            type="filepath",
            file_types=[".json"],
            label="Emotion Analysis",
        ).set_info("Mood probabilities and valence-arousal scores."),
    ]
    build_endpoint(
        model_card=model_card,
        input_components=input_components,
        output_components=output_components,
        process_fn=process_fn,
    )


if __name__ == "__main__":
    demo.queue(default_concurrency_limit=1).launch(show_error=True, pwa=True)