"""Genera el vídeo con los subtítulos incrustados, palabra a palabra. Cada hablante tiene su color, tomado del degradado del logo de subtify, y dentro de cada frase la palabra que se está pronunciando se resalta en el momento exacto que marca su timestamp. Eso hace que un desajuste de tiempos se vea a simple vista. Se usa el formato ASS con etiquetas de karaoke y se incrusta con ffmpeg, que ya forma parte del proyecto: renderizar con Remotion exigiría Node y Chromium, que el Space no tiene. """ import os import subprocess # Degradado del logo de subtify: púrpura -> magenta -> naranja salmón. # El orden alterna tonos para que dos hablantes seguidos no se parezcan. SPEAKER_COLORS = [ "#6C2A8E", # púrpura "#E8734A", # naranja salmón "#B83C6E", # magenta "#4A2A7A", # índigo "#F0A05A", # ámbar "#D94F8C", # rosa ] FONT_NAME = "DejaVu Sans" FONT_SIZE = 42 # Reagrupación de palabras en frases legibles MAX_CHARS = 60 MAX_DURATION = 5.0 MAX_GAP = 0.8 def _hex_to_ass(color, alpha=0): """Convierte #RRGGBB al formato de color de ASS, que es &HAABBGGRR.""" color = color.lstrip("#") r, g, b = int(color[0:2], 16), int(color[2:4], 16), int(color[4:6], 16) return f"&H{alpha:02X}{b:02X}{g:02X}{r:02X}" def _text_color_for(background): """Elige texto claro u oscuro según lo luminoso que sea el fondo.""" color = background.lstrip("#") r, g, b = int(color[0:2], 16), int(color[2:4], 16), int(color[4:6], 16) luminance = (0.299 * r + 0.587 * g + 0.114 * b) / 255 return "#101827" if luminance > 0.6 else "#FFFFFF" def _dim(color, factor=0.45): """Versión atenuada de un color, para las palabras aún no pronunciadas.""" color = color.lstrip("#") r, g, b = int(color[0:2], 16), int(color[2:4], 16), int(color[4:6], 16) return "#%02X%02X%02X" % (int(r * factor), int(g * factor), int(b * factor)) def _timestamp(seconds): """Formato de tiempo de ASS: h:mm:ss.cc""" if seconds < 0: seconds = 0.0 centis = int(round(seconds * 100)) hours, centis = divmod(centis, 360000) minutes, centis = divmod(centis, 6000) secs, centis = divmod(centis, 100) return f"{hours}:{minutes:02d}:{secs:02d}.{centis:02d}" def speaker_order(chunks): """Asigna un índice estable a cada hablante, por orden de aparición.""" order = {} for chunk in chunks: speaker = chunk.get("speaker") or "SPEAKER_00" if speaker not in order: order[speaker] = len(order) return order def group_into_phrases(chunks): """Agrupa las palabras en frases de subtítulo. Una frase se corta cuando cambia el hablante, cuando se hace demasiado larga o cuando hay un silencio apreciable. """ phrases = [] current = None for chunk in chunks: text = (chunk.get("text") or "").strip() if not text: continue start = float(chunk["start"]) end = float(chunk["end"]) speaker = chunk.get("speaker") or "SPEAKER_00" if current is None: current = {"speaker": speaker, "start": start, "end": end, "words": []} candidate_len = sum(len(w["text"]) + 1 for w in current["words"]) + len(text) breaks = ( speaker != current["speaker"] or candidate_len > MAX_CHARS or end - current["start"] > MAX_DURATION or start - current["end"] > MAX_GAP ) if breaks and current["words"]: phrases.append(current) current = {"speaker": speaker, "start": start, "end": end, "words": []} current["words"].append({"text": text, "start": start, "end": end}) current["end"] = end if current and current["words"]: phrases.append(current) return phrases def build_ass(chunks, width=1920, height=1080): """Construye el contenido del fichero ASS con el karaoke por palabra.""" order = speaker_order(chunks) phrases = group_into_phrases(chunks) lines = [ "[Script Info]", "ScriptType: v4.00+", f"PlayResX: {width}", f"PlayResY: {height}", "WrapStyle: 0", "ScaledBorderAndShadow: yes", "", "[V4+ Styles]", "Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, " "OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, " "ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, " "MarginL, MarginR, MarginV, Encoding", ] # Un estilo por hablante. BorderStyle 3 pinta una caja sólida con BackColour, # que es lo que da el fondo de color de cada hablante. for speaker, index in order.items(): background = SPEAKER_COLORS[index % len(SPEAKER_COLORS)] text_color = _text_color_for(background) lines.append( f"Style: S{index},{FONT_NAME},{FONT_SIZE}," f"{_hex_to_ass(text_color)},{_hex_to_ass(_dim(text_color))}," f"{_hex_to_ass(background)},{_hex_to_ass(background)}," f"-1,0,0,0,100,100,0,0,3,6,0,2,60,60,60,1" ) lines += ["", "[Events]", "Format: Layer, Start, End, Style, Name, MarginL, MarginR, " "MarginV, Effect, Text"] for phrase in phrases: index = order[phrase["speaker"]] parts = [] cursor = phrase["start"] for word in phrase["words"]: # Un hueco antes de la palabra se rellena con un karaoke vacío, o el # resaltado se adelantaría respecto al audio gap = int(round((word["start"] - cursor) * 100)) if gap > 0: parts.append(f"{{\\k{gap}}}") duration = max(1, int(round((word["end"] - word["start"]) * 100))) parts.append(f"{{\\k{duration}}}{word['text']} ") cursor = word["end"] text = "".join(parts).strip() lines.append( f"Dialogue: 0,{_timestamp(phrase['start'])},{_timestamp(phrase['end'])}," f"S{index},,0,0,0,,{text}" ) return "\n".join(lines) + "\n" def get_video_size(video_path): """Devuelve (ancho, alto) del vídeo, para que el ASS use su misma resolución.""" result = subprocess.run( ["ffprobe", "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", "-of", "csv=p=0:s=x", video_path], capture_output=True, text=True, check=True, ) width, height = result.stdout.strip().split("x")[:2] return int(width), int(height) def render_subtitled_video(video_path, chunks, output_path, ass_path=None): """Incrusta los subtítulos en el vídeo. Args: video_path: vídeo original. chunks: lista de palabras con start, end, text y speaker. output_path: dónde guardar el vídeo resultante. ass_path: dónde dejar el ASS generado; junto al vídeo si no se indica. Returns: str: ruta del vídeo generado. """ if not chunks: raise ValueError("No hay palabras que subtitular") width, height = get_video_size(video_path) ass_path = ass_path or os.path.splitext(output_path)[0] + ".ass" os.makedirs(os.path.dirname(os.path.abspath(output_path)), exist_ok=True) with open(ass_path, "w", encoding="utf-8") as f: f.write(build_ass(chunks, width, height)) # El filtro subtitles no admite rutas con caracteres sin escapar escaped = ass_path.replace("\\", "\\\\").replace(":", r"\:").replace("'", r"\'") subprocess.run( ["ffmpeg", "-y", "-v", "error", "-i", video_path, "-vf", f"subtitles='{escaped}'", "-c:a", "copy", output_path], check=True, ) return output_path