mesmer-tts / app.js
mesmertech's picture
Deploy static space build
27b1988 verified
Raw
History Blame Contribute Delete
7.95 kB
/**
* Text to Speech — MesmerTools HuggingFace Space.
*
* Static page that calls the public mesmer.tools TTS API directly from the
* visitor's browser, so the 20 req/hour free limit applies per visitor (their
* own IP) rather than through one shared Space backend.
*
* Voices are not hardcoded: they're fetched at startup from the generated
* tts-voices.json on mesmer.tools (canonical source = the main repo). A tiny
* built-in default keeps the space working if that fetch ever fails.
*/
import { createSpace, el } from "./shared/ui.js";
import { callRest, fetchData } from "./shared/api-client.js";
import { DATA, ENDPOINTS, siteUrl } from "./shared/config.js";
/** Minimal fallback so the space still works if the voices JSON can't load. */
const FALLBACK_VOICES = {
maxTextLength: 2000,
defaultQuality: "high",
hq: {
default: "hq_female",
voices: [{ id: "hq_female", label: "Warm Female", gender: "female" }],
},
kokoro: {
default: "af_bella",
voices: [
{ id: "af_bella", label: "Bella", gender: "female", language: "American English" },
],
},
};
const FULL_TOOL_URL = siteUrl(ENDPOINTS.tts.fullToolPath);
const space = createSpace({
toolKey: "tts",
emoji: "🔊",
title: "Free Text to Speech API",
subtitle:
"Convert text into natural AI speech and download the MP3 — a free AI voice generator online. Pick a voice, paste your text, no signup.",
});
/** True if the fetched voices JSON has the shape we need. */
function isValidVoices(d) {
return (
d &&
typeof d === "object" &&
d.hq &&
Array.isArray(d.hq.voices) &&
d.hq.voices.length > 0 &&
d.kokoro &&
Array.isArray(d.kokoro.voices) &&
d.kokoro.voices.length > 0
);
}
async function loadVoices() {
try {
const data = await fetchData(DATA.ttsVoices);
if (isValidVoices(data)) return data;
} catch {
/* fall through to the built-in default */
}
return FALLBACK_VOICES;
}
/** Friendly option label for one voice within a quality tier. */
function voiceOptionLabel(v, tierValue) {
if (tierValue === "low") {
const extra = v.language || v.gender;
return extra ? `${v.label}${extra}` : v.label;
}
const extra = v.description || v.gender;
return extra ? `${v.label}${extra}` : v.label;
}
function buildForm(data) {
const maxLen = Number(data.maxTextLength) > 0 ? Number(data.maxTextLength) : 2000;
// Quality tiers map the two voice buckets to the API's quality values.
const tiers = [
{ value: "high", label: "High quality (AI)", bucket: data.hq || {} },
{ value: "low", label: "Standard (fast)", bucket: data.kokoro || {} },
];
const defaultQuality =
data.defaultQuality === "low" || data.defaultQuality === "high"
? data.defaultQuality
: "high";
// id -> friendly label across both buckets, for the result meta line.
const labelById = new Map();
for (const t of tiers) {
for (const v of t.bucket.voices || []) {
labelById.set(v.id, voiceOptionLabel(v, t.value));
}
}
// --- Text -----------------------------------------------------------
const textarea = el("textarea", {
className: "ms-textarea",
placeholder: "Type or paste the text you want spoken aloud…",
maxlength: String(maxLen),
required: "",
rows: "5",
});
const charCount = el("div", { className: "ms-char-count", text: `0 / ${maxLen}` });
const updateCount = () => {
charCount.textContent = `${textarea.value.length} / ${maxLen}`;
};
textarea.addEventListener("input", updateCount);
const textField = el("div", { className: "ms-field" },
el("label", { for: "tts-text", text: "Text" }),
Object.assign(textarea, { id: "tts-text" }),
charCount,
);
// --- Quality --------------------------------------------------------
const qualitySelect = el("select", { className: "ms-select", id: "tts-quality" },
...tiers.map((t) => el("option", { value: t.value, text: t.label })),
);
qualitySelect.value = defaultQuality;
const qualityField = el("div", { className: "ms-field" },
el("label", { for: "tts-quality", text: "Quality" }),
qualitySelect,
el("span", {
className: "ms-hint",
text: "High quality uses studio AI voices; Standard is faster and lighter.",
}),
);
// --- Voice ----------------------------------------------------------
const voiceSelect = el("select", { className: "ms-select", id: "tts-voice" });
const voiceField = el("div", { className: "ms-field" },
el("label", { for: "tts-voice", text: "Voice" }),
voiceSelect,
);
const currentTier = () => tiers.find((t) => t.value === qualitySelect.value) || tiers[0];
function populateVoices(preferredId) {
const tier = currentTier();
const voices = tier.bucket.voices || [];
voiceSelect.replaceChildren(
...voices.map((v) => el("option", { value: v.id, text: voiceOptionLabel(v, tier.value) })),
);
const wanted =
preferredId && voices.some((v) => v.id === preferredId)
? preferredId
: tier.bucket.default || (voices[0] && voices[0].id);
if (wanted) voiceSelect.value = wanted;
}
qualitySelect.addEventListener("change", () => populateVoices());
populateVoices(currentTier().bucket.default);
// --- Submit + note --------------------------------------------------
const generateBtn = el("button", { className: "ms-btn", type: "submit", text: "Generate speech" });
const apiNote = el("p", { className: "ms-api-note" },
`Free demo: ${ENDPOINTS.tts.freeLimitPerHour}/hour per visitor. Need more, batch, or API access? `,
el("a", { href: FULL_TOOL_URL, target: "_blank", rel: "noopener", text: "Use the full text to speech tool on mesmer.tools →" }),
);
space.form.append(
textField,
el("div", { className: "ms-row" }, qualityField, voiceField),
generateBtn,
apiNote,
);
// --- Result rendering ----------------------------------------------
function renderResult(data) {
const voiceLabel = labelById.get(data.voice) || data.voice;
const parts = [`Voice: ${voiceLabel}`, `Provider: ${data.provider}`];
if (data.cached) parts.push("cached");
const audio = el("audio", { className: "ms-audio", controls: "", src: data.url, preload: "auto" });
const children = [
el("div", { className: "ms-result-frame", style: { padding: "14px" } }, audio),
el("div", { className: "ms-result-actions" },
el("a", {
className: "ms-btn ms-btn-ghost",
href: data.url,
download: "speech.mp3",
target: "_blank",
rel: "noopener",
text: "Download MP3",
}),
),
el("p", { className: "ms-meta", text: parts.join(" · ") }),
];
if (data.fellBack) {
children.push(el("p", { className: "ms-meta ms-fellback",
text: "High-quality was busy — generated with the standard voice instead." }));
}
space.clearOutput();
space.output.append(...children);
// Autoplay is optional — controls are always available if the browser blocks it.
audio.play().catch(() => {});
}
// --- Submit handler -------------------------------------------------
space.form.addEventListener("submit", async (e) => {
e.preventDefault();
const text = textarea.value.trim();
if (!text) {
space.showError("Enter some text to convert to speech.");
textarea.focus();
return;
}
const body = { text, voice: voiceSelect.value, quality: qualitySelect.value };
generateBtn.disabled = true;
space.setLoading(true, "Generating audio…");
try {
const data = await callRest("/api/v1/tts", { method: "POST", body, timeout: 90_000 });
space.clearStatus();
renderResult(data);
} catch (err) {
space.showError(err);
} finally {
generateBtn.disabled = false;
}
});
}
(async function init() {
const data = await loadVoices();
buildForm(data);
})();