scrape / lib /tts.ts
kjbytes's picture
push code
ef4c36f
Raw
History Blame Contribute Delete
5.52 kB
import fs from "fs/promises";
import { execFile } from "child_process";
import { promisify } from "util";
import ffprobeStatic from "ffprobe-static";
import ffmpegStatic from "ffmpeg-static";
import path from "path";
import { enforceScriptLimit } from "@/lib/script-limits";
import { syncCaptionsToDuration } from "@/lib/timing";
import type { Caption } from "@/types";
const DEFAULT_API_BASE = "https://d3evil4-gitre.hf.space";
export interface TtsResult {
duration: number;
captions: Caption[];
}
function getApiBase(): string {
return (process.env.TTS_API_URL || DEFAULT_API_BASE).replace(/\/$/, "");
}
function getApiKey(): string {
const apiKey = process.env.TTS_API_KEY;
if (!apiKey) {
throw new Error("TTS_API_KEY is required. Add it to .env.local");
}
return apiKey;
}
async function getAudioDuration(audioFile: string): Promise<number> {
const execFileAsync = promisify(execFile);
const { stdout } = await execFileAsync(ffprobeStatic.path, [
"-v",
"error",
"-show_entries",
"format=duration",
"-of",
"default=noprint_wrappers=1:nokey=1",
audioFile,
]);
return Number(stdout.trim()) || 30;
}
// ponytail: chatterbox truncates long input (48 words came back as 6.4s of audio),
// so synthesize one sentence at a time — that also gives real per-sentence timings.
const MAX_CHUNK_CHARS = 140;
export function splitIntoChunks(text: string): string[] {
const sentences = text.split(/(?<=[.!?])\s+/).filter(Boolean);
const chunks: string[] = [];
for (const sentence of sentences) {
const last = chunks[chunks.length - 1];
if (last && `${last} ${sentence}`.length <= MAX_CHUNK_CHARS) {
chunks[chunks.length - 1] = `${last} ${sentence}`;
} else if (sentence.length <= MAX_CHUNK_CHARS) {
chunks.push(sentence);
} else {
// a single over-long sentence: break it on commas, then on word count
let buffer = "";
for (const part of sentence.split(/(?<=,)\s+/)) {
if (`${buffer} ${part}`.trim().length <= MAX_CHUNK_CHARS) {
buffer = `${buffer} ${part}`.trim();
} else {
if (buffer) chunks.push(buffer);
buffer = part;
}
}
if (buffer) chunks.push(buffer);
}
}
return chunks.length ? chunks : [text];
}
async function synthesizeChunk(chunk: string, dest: string): Promise<void> {
const apiBase = getApiBase();
const ttsRes = await fetch(`${apiBase}/tts`, {
method: "POST",
headers: { "Content-Type": "application/json", "x-api-key": getApiKey() },
body: JSON.stringify({
text: chunk,
lang: process.env.TTS_LANG || "en",
exaggeration: Number(process.env.TTS_EXAGGERATION ?? 0.5),
// ponytail: chatterbox rushes at 0.5; lower cfg_weight = slower, natural pacing
cfg_weight: Number(process.env.TTS_CFG_WEIGHT ?? 0.3),
}),
});
if (!ttsRes.ok) {
const errorText = await ttsRes.text();
throw new Error(`TTS failed (${ttsRes.status}): ${errorText.slice(0, 200)}`);
}
const { url } = (await ttsRes.json()) as { url?: string };
if (!url) throw new Error("TTS did not return an audio url");
// ponytail: API hands back its own host; reuse our base so https/self-host works
const audioRes = await fetch(`${apiBase}${new URL(url).pathname}`, {
headers: { "x-api-key": getApiKey() },
});
if (!audioRes.ok) {
throw new Error(`TTS audio download failed (${audioRes.status})`);
}
await fs.writeFile(dest, Buffer.from(await audioRes.arrayBuffer()));
}
export async function generateSpeech(
script: string,
outputPath: string,
): Promise<TtsResult> {
const text = enforceScriptLimit(script);
const execFileAsync = promisify(execFile);
const workDir = path.join(path.dirname(outputPath), "tts-chunks");
await fs.mkdir(workDir, { recursive: true });
const chunks = splitIntoChunks(text);
const chunkFiles: string[] = [];
const captions: Caption[] = [];
try {
// ponytail: calibration knob — model pacing varies per voice, stretch without pitch shift
const tempo = Number(process.env.TTS_TEMPO ?? 0.92);
let cursorMs = 0;
for (const [index, chunk] of chunks.entries()) {
const chunkFile = path.join(workDir, `${index}.wav`);
await synthesizeChunk(chunk, chunkFile);
chunkFiles.push(chunkFile);
// real anchor per chunk: measured audio, not a guessed proportional split
const chunkMs = (await getAudioDuration(chunkFile)) * 1000 / tempo;
for (const caption of syncCaptionsToDuration(chunk, chunkMs)) {
captions.push({
...caption,
startMs: Math.round(cursorMs + caption.startMs),
endMs: Math.round(cursorMs + caption.endMs),
});
}
cursorMs += chunkMs;
}
const listFile = path.join(workDir, "list.txt");
await fs.writeFile(
listFile,
chunkFiles.map((f) => `file '${f.replace(/'/g, "'\\''")}'`).join("\n"),
);
await execFileAsync(ffmpegStatic as string, [
"-y",
"-f", "concat",
"-safe", "0",
"-i", listFile,
...(tempo !== 1 ? ["-filter:a", `atempo=${tempo}`] : []),
// ponytail: codec must match the container — outputPath is .mp3, PCM won't mux
...(outputPath.endsWith(".wav")
? ["-c:a", "pcm_s16le"]
: ["-c:a", "libmp3lame", "-b:a", "192k"]),
outputPath,
]);
} finally {
await fs.rm(workDir, { recursive: true, force: true });
}
const duration = await getAudioDuration(outputPath);
return { duration, captions };
}