subtify / subtitle_render.py
Maximofn's picture
Idioma auto y subtitulos incrustados palabra a palabra con color por hablante
f12d1d9 verified
Raw
History Blame Contribute Delete
7.71 kB
"""Genera el vídeo con los subtítulos incrustados, palabra a palabra.
Cada hablante tiene su color, tomado del degradado del logo de subtify, y dentro
de cada frase la palabra que se está pronunciando se resalta en el momento exacto
que marca su timestamp. Eso hace que un desajuste de tiempos se vea a simple vista.
Se usa el formato ASS con etiquetas de karaoke y se incrusta con ffmpeg, que ya
forma parte del proyecto: renderizar con Remotion exigiría Node y Chromium, que el
Space no tiene.
"""
import os
import subprocess
# Degradado del logo de subtify: púrpura -> magenta -> naranja salmón.
# El orden alterna tonos para que dos hablantes seguidos no se parezcan.
SPEAKER_COLORS = [
"#6C2A8E", # púrpura
"#E8734A", # naranja salmón
"#B83C6E", # magenta
"#4A2A7A", # índigo
"#F0A05A", # ámbar
"#D94F8C", # rosa
]
FONT_NAME = "DejaVu Sans"
FONT_SIZE = 42
# Reagrupación de palabras en frases legibles
MAX_CHARS = 60
MAX_DURATION = 5.0
MAX_GAP = 0.8
def _hex_to_ass(color, alpha=0):
"""Convierte #RRGGBB al formato de color de ASS, que es &HAABBGGRR."""
color = color.lstrip("#")
r, g, b = int(color[0:2], 16), int(color[2:4], 16), int(color[4:6], 16)
return f"&H{alpha:02X}{b:02X}{g:02X}{r:02X}"
def _text_color_for(background):
"""Elige texto claro u oscuro según lo luminoso que sea el fondo."""
color = background.lstrip("#")
r, g, b = int(color[0:2], 16), int(color[2:4], 16), int(color[4:6], 16)
luminance = (0.299 * r + 0.587 * g + 0.114 * b) / 255
return "#101827" if luminance > 0.6 else "#FFFFFF"
def _dim(color, factor=0.45):
"""Versión atenuada de un color, para las palabras aún no pronunciadas."""
color = color.lstrip("#")
r, g, b = int(color[0:2], 16), int(color[2:4], 16), int(color[4:6], 16)
return "#%02X%02X%02X" % (int(r * factor), int(g * factor), int(b * factor))
def _timestamp(seconds):
"""Formato de tiempo de ASS: h:mm:ss.cc"""
if seconds < 0:
seconds = 0.0
centis = int(round(seconds * 100))
hours, centis = divmod(centis, 360000)
minutes, centis = divmod(centis, 6000)
secs, centis = divmod(centis, 100)
return f"{hours}:{minutes:02d}:{secs:02d}.{centis:02d}"
def speaker_order(chunks):
"""Asigna un índice estable a cada hablante, por orden de aparición."""
order = {}
for chunk in chunks:
speaker = chunk.get("speaker") or "SPEAKER_00"
if speaker not in order:
order[speaker] = len(order)
return order
def group_into_phrases(chunks):
"""Agrupa las palabras en frases de subtítulo.
Una frase se corta cuando cambia el hablante, cuando se hace demasiado larga
o cuando hay un silencio apreciable.
"""
phrases = []
current = None
for chunk in chunks:
text = (chunk.get("text") or "").strip()
if not text:
continue
start = float(chunk["start"])
end = float(chunk["end"])
speaker = chunk.get("speaker") or "SPEAKER_00"
if current is None:
current = {"speaker": speaker, "start": start, "end": end, "words": []}
candidate_len = sum(len(w["text"]) + 1 for w in current["words"]) + len(text)
breaks = (
speaker != current["speaker"]
or candidate_len > MAX_CHARS
or end - current["start"] > MAX_DURATION
or start - current["end"] > MAX_GAP
)
if breaks and current["words"]:
phrases.append(current)
current = {"speaker": speaker, "start": start, "end": end, "words": []}
current["words"].append({"text": text, "start": start, "end": end})
current["end"] = end
if current and current["words"]:
phrases.append(current)
return phrases
def build_ass(chunks, width=1920, height=1080):
"""Construye el contenido del fichero ASS con el karaoke por palabra."""
order = speaker_order(chunks)
phrases = group_into_phrases(chunks)
lines = [
"[Script Info]",
"ScriptType: v4.00+",
f"PlayResX: {width}",
f"PlayResY: {height}",
"WrapStyle: 0",
"ScaledBorderAndShadow: yes",
"",
"[V4+ Styles]",
"Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, "
"OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, "
"ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, "
"MarginL, MarginR, MarginV, Encoding",
]
# Un estilo por hablante. BorderStyle 3 pinta una caja sólida con BackColour,
# que es lo que da el fondo de color de cada hablante.
for speaker, index in order.items():
background = SPEAKER_COLORS[index % len(SPEAKER_COLORS)]
text_color = _text_color_for(background)
lines.append(
f"Style: S{index},{FONT_NAME},{FONT_SIZE},"
f"{_hex_to_ass(text_color)},{_hex_to_ass(_dim(text_color))},"
f"{_hex_to_ass(background)},{_hex_to_ass(background)},"
f"-1,0,0,0,100,100,0,0,3,6,0,2,60,60,60,1"
)
lines += ["", "[Events]",
"Format: Layer, Start, End, Style, Name, MarginL, MarginR, "
"MarginV, Effect, Text"]
for phrase in phrases:
index = order[phrase["speaker"]]
parts = []
cursor = phrase["start"]
for word in phrase["words"]:
# Un hueco antes de la palabra se rellena con un karaoke vacío, o el
# resaltado se adelantaría respecto al audio
gap = int(round((word["start"] - cursor) * 100))
if gap > 0:
parts.append(f"{{\\k{gap}}}")
duration = max(1, int(round((word["end"] - word["start"]) * 100)))
parts.append(f"{{\\k{duration}}}{word['text']} ")
cursor = word["end"]
text = "".join(parts).strip()
lines.append(
f"Dialogue: 0,{_timestamp(phrase['start'])},{_timestamp(phrase['end'])},"
f"S{index},,0,0,0,,{text}"
)
return "\n".join(lines) + "\n"
def get_video_size(video_path):
"""Devuelve (ancho, alto) del vídeo, para que el ASS use su misma resolución."""
result = subprocess.run(
["ffprobe", "-v", "error", "-select_streams", "v:0",
"-show_entries", "stream=width,height", "-of", "csv=p=0:s=x", video_path],
capture_output=True, text=True, check=True,
)
width, height = result.stdout.strip().split("x")[:2]
return int(width), int(height)
def render_subtitled_video(video_path, chunks, output_path, ass_path=None):
"""Incrusta los subtítulos en el vídeo.
Args:
video_path: vídeo original.
chunks: lista de palabras con start, end, text y speaker.
output_path: dónde guardar el vídeo resultante.
ass_path: dónde dejar el ASS generado; junto al vídeo si no se indica.
Returns:
str: ruta del vídeo generado.
"""
if not chunks:
raise ValueError("No hay palabras que subtitular")
width, height = get_video_size(video_path)
ass_path = ass_path or os.path.splitext(output_path)[0] + ".ass"
os.makedirs(os.path.dirname(os.path.abspath(output_path)), exist_ok=True)
with open(ass_path, "w", encoding="utf-8") as f:
f.write(build_ass(chunks, width, height))
# El filtro subtitles no admite rutas con caracteres sin escapar
escaped = ass_path.replace("\\", "\\\\").replace(":", r"\:").replace("'", r"\'")
subprocess.run(
["ffmpeg", "-y", "-v", "error", "-i", video_path,
"-vf", f"subtitles='{escaped}'",
"-c:a", "copy", output_path],
check=True,
)
return output_path