File size: 9,987 Bytes
e93dbf9
98765df
 
611c270
e93dbf9
828703c
611c270
e93dbf9
611c270
e93dbf9
 
98765df
 
e93dbf9
 
98765df
 
 
 
 
051ce4d
e93dbf9
 
 
611c270
98765df
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
611c270
98765df
 
 
 
 
 
 
 
 
 
 
 
611c270
 
 
e93dbf9
 
 
 
 
 
 
 
 
828703c
 
 
e93dbf9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
611c270
 
98765df
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
611c270
 
 
e93dbf9
 
98765df
e93dbf9
611c270
98765df
611c270
 
 
 
e93dbf9
611c270
98765df
051ce4d
98765df
611c270
2fce972
98765df
611c270
98765df
611c270
 
98765df
ab5ab1b
 
 
98765df
 
611c270
 
e93dbf9
611c270
e93dbf9
98765df
 
 
611c270
 
 
 
 
 
 
 
 
 
 
 
 
 
a810aea
98765df
 
 
 
 
 
611c270
 
 
 
98765df
611c270
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
import os
import uuid
import subprocess
import gradio as gr
import requests
from faster_whisper import WhisperModel

API_KEY = os.environ.get("SILICONFLOW_API_KEY", "")

ASR_URL = "https://api.siliconflow.cn/v1/audio/transcriptions"
CHAT_URL = "https://api.siliconflow.cn/v1/chat/completions"
TTS_URL = "https://api.siliconflow.cn/v1/audio/speech"

ASR_MODEL = "FunAudioLLM/SenseVoiceSmall"
MT_MODEL = "tencent/Hunyuan-MT-7B"
TTS_MODEL = "FunAudioLLM/CosyVoice2-0.5B"

# 音频文件统一存到这个目录,用唯一文件名,实现"可重复回放历史"
AUDIO_DIR = "/tmp/chuanshengtong_audio"
os.makedirs(AUDIO_DIR, exist_ok=True)

# 只有法语识别本地跑,其他都走硅基流动,兼顾速度和成本
FRENCH_ASR_MODEL_SIZE = "medium"
french_asr_model = WhisperModel(FRENCH_ASR_MODEL_SIZE, device="cpu", compute_type="int8")

# ---------------------------------------------------------------------------
# Piper 本地法语 TTS:硅基流动的 CosyVoice2 不支持法语合成,所以法语这一路
# 用本地 Piper 兜底。HF Space 容器每次重建都是空的,所以在启动时自动下载模型,
# 不需要手动登录服务器操作。
# ---------------------------------------------------------------------------
PIPER_BIN = "piper"  # pip install piper-tts 后,命令行工具会自动加入 PATH

PIPER_MODEL_DIR = "piper_models"
os.makedirs(PIPER_MODEL_DIR, exist_ok=True)

PIPER_MODEL_FR = os.path.join(PIPER_MODEL_DIR, "fr_FR-siwis-medium.onnx")
PIPER_MODEL_FR_JSON = os.path.join(PIPER_MODEL_DIR, "fr_FR-siwis-medium.onnx.json")

PIPER_MODEL_URL = "https://huggingface.co/rhasspy/piper-voices/resolve/main/fr/fr_FR/siwis/medium/fr_FR-siwis-medium.onnx"
PIPER_MODEL_JSON_URL = "https://huggingface.co/rhasspy/piper-voices/resolve/main/fr/fr_FR/siwis/medium/fr_FR-siwis-medium.onnx.json"


def ensure_piper_model():
    """启动时检查法语模型文件是否存在,不存在就自动下载。"""
    try:
        if not os.path.exists(PIPER_MODEL_FR):
            print("正在下载 Piper 法语模型(onnx)...")
            r = requests.get(PIPER_MODEL_URL, timeout=120)
            r.raise_for_status()
            with open(PIPER_MODEL_FR, "wb") as f:
                f.write(r.content)

        if not os.path.exists(PIPER_MODEL_FR_JSON):
            print("正在下载 Piper 法语模型配置(json)...")
            r = requests.get(PIPER_MODEL_JSON_URL, timeout=60)
            r.raise_for_status()
            with open(PIPER_MODEL_FR_JSON, "wb") as f:
                f.write(r.content)

        print("Piper 法语模型已就绪")
    except Exception as e:
        print(f"⚠️ Piper 法语模型下载失败:{e}(法语合成方向届时会报错,其他方向不受影响)")


ensure_piper_model()

# ---------------------------------------------------------------------------
# 翻译方向配置
# asr : 识别引擎  "siliconflow" | "local_french"
# tts : 合成引擎  "siliconflow" | "piper_fr"
# voice: 硅基流动 CosyVoice2 的预置音色(走 piper_fr 时无用,填 None)
# ---------------------------------------------------------------------------
LANG_MAP = {
    "英语 → 中文": {"src": "English", "tgt": "Chinese", "src_tag": "EN", "tgt_tag": "中",
                    "asr": "siliconflow", "tts": "siliconflow", "voice": "FunAudioLLM/CosyVoice2-0.5B:alex"},
    "汉语 → 英文": {"src": "Chinese", "tgt": "English", "src_tag": "中", "tgt_tag": "EN",
                    "asr": "siliconflow", "tts": "siliconflow", "voice": "FunAudioLLM/CosyVoice2-0.5B:david"},
    "汉语 → 法文": {"src": "Chinese", "tgt": "French", "src_tag": "中", "tgt_tag": "FR",
                    "asr": "siliconflow", "tts": "piper_fr", "voice": None},
    "法语 → 中文": {"src": "French", "tgt": "Chinese", "src_tag": "FR", "tgt_tag": "中",
                    "asr": "local_french", "tts": "siliconflow", "voice": "FunAudioLLM/CosyVoice2-0.5B:alex"},
    "英语 → 法文": {"src": "English", "tgt": "French", "src_tag": "EN", "tgt_tag": "FR",
                    "asr": "siliconflow", "tts": "piper_fr", "voice": None},
    "法语 → 英文": {"src": "French", "tgt": "English", "src_tag": "FR", "tgt_tag": "EN",
                    "asr": "local_french", "tts": "siliconflow", "voice": "FunAudioLLM/CosyVoice2-0.5B:david"},
}


def transcribe_siliconflow(audio_path):
    with open(audio_path, "rb") as f:
        files = {"file": (os.path.basename(audio_path), f, "audio/wav")}
        data = {"model": ASR_MODEL}
        headers = {"Authorization": "Bearer " + API_KEY}
        r = requests.post(ASR_URL, headers=headers, data=data, files=files, timeout=60)
    r.raise_for_status()
    text = r.json().get("text", "").strip()
    text = text.replace("<|", "").replace("|>", "")
    return text


def transcribe_french_local(audio_path):
    segments, _info = french_asr_model.transcribe(
        audio_path,
        language="fr",
        task="transcribe",
        vad_filter=True,
    )
    return "".join(seg.text for seg in segments).strip()


def transcribe(audio_path, engine):
    if engine == "local_french":
        return transcribe_french_local(audio_path)
    return transcribe_siliconflow(audio_path)


def translate(text, src, tgt):
    headers = {"Authorization": "Bearer " + API_KEY, "Content-Type": "application/json"}
    payload = {
        "model": MT_MODEL,
        "messages": [
            {
                "role": "system",
                "content": f"You are a professional translator. Translate the user text from {src} to {tgt}. "
                           f"Output ONLY the translated text, no explanations, no quotes, no pinyin.",
            },
            {"role": "user", "content": text},
        ],
        "temperature": 0.3,
        "max_tokens": 512,
    }
    r = requests.post(CHAT_URL, headers=headers, json=payload, timeout=60)
    r.raise_for_status()
    return r.json()["choices"][0]["message"]["content"].strip()


def speak_siliconflow(text, voice, out_path):
    """硅基流动 CosyVoice2 合成,输出 mp3"""
    headers = {"Authorization": "Bearer " + API_KEY, "Content-Type": "application/json"}
    payload = {
        "model": TTS_MODEL,
        "input": text,
        "voice": voice,
        "response_format": "mp3",
    }
    r = requests.post(TTS_URL, headers=headers, json=payload, timeout=60)
    r.raise_for_status()
    with open(out_path, "wb") as f:
        f.write(r.content)
    return out_path


def speak_piper_french(text, out_path):
    """本地 Piper 合成法语语音,输出 wav"""
    if not os.path.exists(PIPER_MODEL_FR):
        raise RuntimeError(f"未找到 Piper 法语模型文件:{PIPER_MODEL_FR},请检查启动日志中的下载是否成功")
    cmd = [PIPER_BIN, "--model", PIPER_MODEL_FR, "--output_file", out_path]
    proc = subprocess.run(cmd, input=text.encode("utf-8"), capture_output=True)
    if proc.returncode != 0:
        err = proc.stderr.decode("utf-8", errors="ignore")
        raise RuntimeError(f"Piper 合成失败:{err}")
    return out_path


def speak(text, cfg):
    """按翻译方向配置,选择硅基流动或本地 Piper 合成语音;文件名唯一,支持历史回放"""
    uid = uuid.uuid4().hex[:8]
    if cfg["tts"] == "piper_fr":
        out_path = os.path.join(AUDIO_DIR, f"{uid}.wav")
        return speak_piper_french(text, out_path)
    else:
        out_path = os.path.join(AUDIO_DIR, f"{uid}.mp3")
        return speak_siliconflow(text, cfg["voice"], out_path)


def process(audio_path, mode_label, history):
    history = history or []

    if not API_KEY:
        history.append({"role": "assistant", "content": "⚠️ 服务器未配置 SILICONFLOW_API_KEY,请检查 Space 的 Settings -> Repository secrets"})
        return history, None, None

    if audio_path is None:
        return history, None, None

    cfg = LANG_MAP[mode_label]

    try:
        text = transcribe(audio_path, cfg["asr"])
        if not text:
            return history, None, None
        trans = translate(text, cfg["src"], cfg["tgt"])
        tts_path = speak(trans, cfg)
    except Exception as e:
        history.append({"role": "assistant", "content": f"⚠️ 出错了:{e}"})
        return history, None, None

    # 文字记录
    history.append({"role": "user", "content": f"[{cfg['src_tag']}] {text}"})
    history.append({"role": "assistant", "content": f"[{cfg['tgt_tag']}] {trans}"})
    # 把这句译文的音频也塞进聊天记录里,实现"点开历史消息随时重听"
    # 注意:新版 Gradio Chatbot 用 messages 格式,文件消息要写成 {"path": ...} 字典,
    # 而不是旧版 tuples 格式的 (path,) 元组,否则会报 Invalid message 错误
    history.append({"role": "assistant", "content": {"path": tts_path}})

    return history, None, tts_path


with gr.Blocks(title="传声筒") as demo:
    gr.Markdown(
        "## 🎙️ 传声筒 · 实时语音传译\n"
        "点麦克风开始录音,说一句完整的话,再点一次停止,自动识别+翻译+朗读译文。\n"
        "识别与翻译主要使用硅基流动,法语识别、法语合成使用本地方案补充。\n"
        "每句译文的语音会自动播放一遍,也会留在下方记录里,随时可以点开重听。"
    )

    mode = gr.Radio(
        choices=list(LANG_MAP.keys()),
        value="英语 → 中文",
        label="翻译方向",
    )

    audio_in = gr.Audio(
        sources=["microphone"],
        type="filepath",
        label="点击录音,再点一次停止",
    )

    chatbot = gr.Chatbot(label="传译记录(含可重听音频)", height=460)

    tts_out = gr.Audio(
        label="最新译文语音(当前这一句,可反复点播放)",
        autoplay=True,
        interactive=False,
    )

    audio_in.stop_recording(
        fn=process,
        inputs=[audio_in, mode, chatbot],
        outputs=[chatbot, audio_in, tts_out],
    )

demo.launch()