teddybear082 commited on
Commit
5aa947f
·
2 Parent(s): f459c16147e805

Merge pull request #11 from teddybear082/fix_opus_output_support

Browse files
README.md CHANGED
@@ -154,7 +154,7 @@ with client.audio.speech.with_streaming_response.create(
154
  | `model` | string | No | - | Ignored (for OpenAI compatibility) |
155
  | `input` | string | Yes | - | Text to synthesize |
156
  | `voice` | string | No | `alba` | Voice ID (see `/v1/voices`) |
157
- | `response_format` | string | No | `mp3` | Output format: `mp3`, `wav`, `pcm`, `opus`, `flac` |
158
  | `stream` | boolean | No | `false` | Enable streaming response |
159
 
160
  ## Custom Voices
 
154
  | `model` | string | No | - | Ignored (for OpenAI compatibility) |
155
  | `input` | string | Yes | - | Text to synthesize |
156
  | `voice` | string | No | `alba` | Voice ID (see `/v1/voices`) |
157
+ | `response_format` | string | No | `mp3` | Output format: `mp3`, `wav`, `pcm`, `opus`, `aac`, `flac` |
158
  | `stream` | boolean | No | `false` | Enable streaming response |
159
 
160
  ## Custom Voices
app/services/audio.py CHANGED
@@ -63,12 +63,34 @@ def convert_audio(
63
  if audio_tensor.dim() == 1:
64
  audio_tensor = audio_tensor.unsqueeze(0)
65
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
66
  try:
67
- torchaudio.save(buffer, audio_tensor, sample_rate, format=target_format)
68
  buffer.seek(0)
69
  return buffer
70
  except Exception as e:
71
- logger.error(f'Error converting audio to {target_format}: {e}')
 
 
72
  raise
73
 
74
 
 
63
  if audio_tensor.dim() == 1:
64
  audio_tensor = audio_tensor.unsqueeze(0)
65
 
66
+ # Handle PCM raw bytes (no container)
67
+ if target_format == 'pcm':
68
+ try:
69
+ pcm_bytes = tensor_to_pcm_bytes(audio_tensor)
70
+ buffer.write(pcm_bytes)
71
+ buffer.seek(0)
72
+ return buffer
73
+ except Exception as e:
74
+ logger.error(f'Error converting audio to PCM: {e}')
75
+ raise
76
+
77
+ # Map OpenAI format names to torchaudio/backend supported format names
78
+ # torchaudio uses 'ogg' as the container for 'opus'
79
+ # 'aac' usually requires 'adts' or 'm4a'
80
+ actual_format = target_format
81
+ if actual_format == 'opus':
82
+ actual_format = 'ogg'
83
+ elif actual_format == 'aac':
84
+ actual_format = 'adts'
85
+
86
  try:
87
+ torchaudio.save(buffer, audio_tensor, sample_rate, format=actual_format)
88
  buffer.seek(0)
89
  return buffer
90
  except Exception as e:
91
+ logger.error(
92
+ f'Error converting audio to {target_format} (backend format: {actual_format}): {e}'
93
+ )
94
  raise
95
 
96
 
pyproject.toml CHANGED
@@ -4,7 +4,7 @@ license = {text = "MIT"}
4
  name = "pocket-tts-openai-server"
5
  readme = "README.md"
6
  requires-python = ">=3.10"
7
- version = "2.5.1"
8
 
9
  dependencies = [
10
  "flask>=3.0.0",
@@ -12,6 +12,7 @@ dependencies = [
12
  "pocket-tts>=2.0.0",
13
  "torch>=2.0.0,<=2.8.0",
14
  "torchaudio>=2.0.0,<=2.8.0",
 
15
  "scipy>=1.10.0",
16
  "numpy>=1.24.0",
17
  "soundfile>=0.12.0",
 
4
  name = "pocket-tts-openai-server"
5
  readme = "README.md"
6
  requires-python = ">=3.10"
7
+ version = "2.5.2"
8
 
9
  dependencies = [
10
  "flask>=3.0.0",
 
12
  "pocket-tts>=2.0.0",
13
  "torch>=2.0.0,<=2.8.0",
14
  "torchaudio>=2.0.0,<=2.8.0",
15
+ "torchao==0.13.0",
16
  "scipy>=1.10.0",
17
  "numpy>=1.24.0",
18
  "soundfile>=0.12.0",
requirements.txt CHANGED
@@ -9,6 +9,7 @@ pocket-tts>=2.0.0
9
  # Audio processing
10
  torch>=2.0.0,<=2.8.0
11
  torchaudio>=2.0.0,<=2.8.0
 
12
  scipy>=1.10.0
13
  numpy>=1.24.0
14
  soundfile>=0.12.0
 
9
  # Audio processing
10
  torch>=2.0.0,<=2.8.0
11
  torchaudio>=2.0.0,<=2.8.0
12
+ torchao==0.13.0
13
  scipy>=1.10.0
14
  numpy>=1.24.0
15
  soundfile>=0.12.0
static/js/app.js CHANGED
@@ -43,6 +43,8 @@ document.addEventListener('DOMContentLoaded', async () => {
43
  if (fmt === 'mp3') {
44
  infoLabel.textContent =
45
  'Streaming is not available for MP3 (Server limitation). A full file will be generated and played.';
 
 
46
  } else {
47
  infoLabel.textContent = 'Streaming is not available for this format.';
48
  }
 
43
  if (fmt === 'mp3') {
44
  infoLabel.textContent =
45
  'Streaming is not available for MP3 (Server limitation). A full file will be generated and played.';
46
+ } else if (['opus', 'aac', 'flac'].includes(fmt)) {
47
+ infoLabel.textContent = `Streaming is not available for ${fmt.toUpperCase()}. A full file will be generated and played.`;
48
  } else {
49
  infoLabel.textContent = 'Streaming is not available for this format.';
50
  }
templates/index.html CHANGED
@@ -126,6 +126,9 @@ Hello! I am Pocket TTS, running locally on your machine. I can clone voices and
126
  "
127
  >
128
  <option value="mp3">MP3</option>
 
 
 
129
  <option value="wav" selected>WAV</option>
130
  <option value="pcm">PCM (Raw)</option>
131
  </select>
@@ -272,8 +275,9 @@ Hello! I am Pocket TTS, running locally on your machine. I can clone voices and
272
  <td></td>
273
  <td><code>mp3</code></td>
274
  <td>
275
- Audio format: <code>mp3</code>, <code>wav</code>,
276
- <code>pcm</code>, <code>opus</code>, <code>flac</code>
 
277
  </td>
278
  </tr>
279
  <tr>
 
126
  "
127
  >
128
  <option value="mp3">MP3</option>
129
+ <option value="opus">Opus</option>
130
+ <option value="aac">AAC</option>
131
+ <option value="flac">FLAC</option>
132
  <option value="wav" selected>WAV</option>
133
  <option value="pcm">PCM (Raw)</option>
134
  </select>
 
275
  <td></td>
276
  <td><code>mp3</code></td>
277
  <td>
278
+ Audio format: <code>mp3</code>, <code>opus</code>,
279
+ <code>aac</code>, <code>flac</code>, <code>wav</code>,
280
+ <code>pcm</code>
281
  </td>
282
  </tr>
283
  <tr>