message / plugins /video /core.py
hunian
refactor(plugins): 插件短名并统一 MCP tool 为 {plugin}-{tool}
cc826a1
Raw
History Blame Contribute Delete
5.87 kB
import base64
import logging
import os
import shutil
import subprocess
import tempfile
from pathlib import Path
from plugins.audio.core import get_transcription_service
logger = logging.getLogger(__name__)
class VideoAudioExtractorService:
"""视频音频提取和转录服务。"""
# 常见视频格式后缀
SUPPORTED_VIDEO_SUFFIXES = {".mp4", ".avi", ".mov", ".mkv", ".webm", ".flv", ".wmv", ".m4v", ".mpg", ".mpeg"}
def extract_audio(self, video_path: str, output_path: str | None = None) -> dict:
"""从视频文件中提取 WAV 音频(16kHz 单声道)。"""
if shutil.which("ffmpeg") is None:
return {
"success": False,
"audio_path": None,
"error": "未找到 ffmpeg,请在运行环境中安装 ffmpeg",
}
source = Path(video_path)
if not source.exists():
return {
"success": False,
"audio_path": None,
"error": f"视频文件不存在: {video_path}",
}
# 获取视频信息(时长等)用于更好的错误提示
video_info = self._probe_video(source)
target = Path(output_path) if output_path else Path(tempfile.mkstemp(suffix=".wav")[1])
command = [
"ffmpeg",
"-y",
"-i",
str(source),
"-vn", # 丢弃视频流
"-acodec",
"pcm_s16le", # 16-bit PCM
"-ar",
"16000", # 16kHz 采样率(Whisper 推荐)
"-ac",
"1", # 单声道
str(target),
]
logger.info(f"开始提取音频: {source.name} (时长: {video_info.get('duration', '未知')})")
try:
result = subprocess.run(
command,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
timeout=300,
)
if result.returncode != 0:
target.unlink(missing_ok=True)
# 提取关键错误信息
error_lines = [l for l in result.stderr.split("\n") if "Error" in l or "Invalid" in l]
error_msg = "\n".join(error_lines) if error_lines else result.stderr[-500:]
logger.error(f"ffmpeg 提取音频失败: {error_msg}")
return {
"success": False,
"audio_path": None,
"error": f"音频提取失败: {error_msg}",
}
logger.info(f"音频提取完成: {target.stat().st_size / 1024:.1f}KB")
return {
"success": True,
"audio_path": str(target),
"error": None,
}
except subprocess.TimeoutExpired:
target.unlink(missing_ok=True)
return {
"success": False,
"audio_path": None,
"error": "音频提取超时(超过5分钟),请检查视频文件大小",
}
def _probe_video(self, source: Path) -> dict:
"""获取视频基本信息。"""
try:
result = subprocess.run(
["ffprobe", "-v", "quiet", "-print_format", "json", "-show_format", str(source)],
capture_output=True,
text=True,
timeout=30,
)
if result.returncode == 0:
import json
info = json.loads(result.stdout)
fmt = info.get("format", {})
return {
"duration": fmt.get("duration", "未知"),
"size_mb": round(int(fmt.get("size", 0)) / (1024 * 1024), 1),
}
except Exception:
pass
return {}
def transcribe_video_file(self, video_path: str, language: str | None = None) -> dict:
"""从视频提取音频并转录文本。"""
extraction = self.extract_audio(video_path)
if not extraction["success"]:
return {
"success": False,
"text": "",
"audio_path": extraction.get("audio_path"),
"error": extraction["error"],
}
audio_path = extraction["audio_path"]
try:
transcription = get_transcription_service().transcribe_file(audio_path, language)
return {
"success": transcription["success"],
"text": transcription.get("text", ""),
"audio_path": audio_path,
"error": transcription.get("error"),
}
finally:
Path(audio_path).unlink(missing_ok=True)
def transcribe_video_base64(
self,
video_base64: str,
suffix: str = ".mp4",
language: str | None = None,
) -> dict:
"""从 Base64 视频提取音频并转录文本。"""
try:
video_bytes = base64.b64decode(video_base64)
except Exception as e:
return {
"success": False,
"text": "",
"audio_path": None,
"error": f"Base64 解码失败: {e}",
}
with tempfile.NamedTemporaryFile(delete=False, suffix=suffix) as tmp_file:
tmp_file.write(video_bytes)
tmp_path = tmp_file.name
try:
return self.transcribe_video_file(tmp_path, language)
finally:
Path(tmp_path).unlink(missing_ok=True)
_video_audio_service: VideoAudioExtractorService | None = None
def get_video_audio_service() -> VideoAudioExtractorService:
global _video_audio_service
if _video_audio_service is None:
_video_audio_service = VideoAudioExtractorService()
return _video_audio_service