| """Genera el vídeo con los subtítulos incrustados, palabra a palabra. |
| |
| Cada hablante tiene su color, tomado del degradado del logo de subtify, y dentro |
| de cada frase la palabra que se está pronunciando se resalta en el momento exacto |
| que marca su timestamp. Eso hace que un desajuste de tiempos se vea a simple vista. |
| |
| Se usa el formato ASS con etiquetas de karaoke y se incrusta con ffmpeg, que ya |
| forma parte del proyecto: renderizar con Remotion exigiría Node y Chromium, que el |
| Space no tiene. |
| """ |
|
|
| import os |
| import subprocess |
|
|
| |
| |
| SPEAKER_COLORS = [ |
| "#6C2A8E", |
| "#E8734A", |
| "#B83C6E", |
| "#4A2A7A", |
| "#F0A05A", |
| "#D94F8C", |
| ] |
|
|
| FONT_NAME = "DejaVu Sans" |
| FONT_SIZE = 42 |
|
|
| |
| MAX_CHARS = 60 |
| MAX_DURATION = 5.0 |
| MAX_GAP = 0.8 |
|
|
|
|
| def _hex_to_ass(color, alpha=0): |
| """Convierte #RRGGBB al formato de color de ASS, que es &HAABBGGRR.""" |
| color = color.lstrip("#") |
| r, g, b = int(color[0:2], 16), int(color[2:4], 16), int(color[4:6], 16) |
| return f"&H{alpha:02X}{b:02X}{g:02X}{r:02X}" |
|
|
|
|
| def _text_color_for(background): |
| """Elige texto claro u oscuro según lo luminoso que sea el fondo.""" |
| color = background.lstrip("#") |
| r, g, b = int(color[0:2], 16), int(color[2:4], 16), int(color[4:6], 16) |
| luminance = (0.299 * r + 0.587 * g + 0.114 * b) / 255 |
| return "#101827" if luminance > 0.6 else "#FFFFFF" |
|
|
|
|
| def _dim(color, factor=0.45): |
| """Versión atenuada de un color, para las palabras aún no pronunciadas.""" |
| color = color.lstrip("#") |
| r, g, b = int(color[0:2], 16), int(color[2:4], 16), int(color[4:6], 16) |
| return "#%02X%02X%02X" % (int(r * factor), int(g * factor), int(b * factor)) |
|
|
|
|
| def _timestamp(seconds): |
| """Formato de tiempo de ASS: h:mm:ss.cc""" |
| if seconds < 0: |
| seconds = 0.0 |
| centis = int(round(seconds * 100)) |
| hours, centis = divmod(centis, 360000) |
| minutes, centis = divmod(centis, 6000) |
| secs, centis = divmod(centis, 100) |
| return f"{hours}:{minutes:02d}:{secs:02d}.{centis:02d}" |
|
|
|
|
| def speaker_order(chunks): |
| """Asigna un índice estable a cada hablante, por orden de aparición.""" |
| order = {} |
| for chunk in chunks: |
| speaker = chunk.get("speaker") or "SPEAKER_00" |
| if speaker not in order: |
| order[speaker] = len(order) |
| return order |
|
|
|
|
| def group_into_phrases(chunks): |
| """Agrupa las palabras en frases de subtítulo. |
| |
| Una frase se corta cuando cambia el hablante, cuando se hace demasiado larga |
| o cuando hay un silencio apreciable. |
| """ |
| phrases = [] |
| current = None |
|
|
| for chunk in chunks: |
| text = (chunk.get("text") or "").strip() |
| if not text: |
| continue |
|
|
| start = float(chunk["start"]) |
| end = float(chunk["end"]) |
| speaker = chunk.get("speaker") or "SPEAKER_00" |
|
|
| if current is None: |
| current = {"speaker": speaker, "start": start, "end": end, "words": []} |
|
|
| candidate_len = sum(len(w["text"]) + 1 for w in current["words"]) + len(text) |
| breaks = ( |
| speaker != current["speaker"] |
| or candidate_len > MAX_CHARS |
| or end - current["start"] > MAX_DURATION |
| or start - current["end"] > MAX_GAP |
| ) |
|
|
| if breaks and current["words"]: |
| phrases.append(current) |
| current = {"speaker": speaker, "start": start, "end": end, "words": []} |
|
|
| current["words"].append({"text": text, "start": start, "end": end}) |
| current["end"] = end |
|
|
| if current and current["words"]: |
| phrases.append(current) |
|
|
| return phrases |
|
|
|
|
| def build_ass(chunks, width=1920, height=1080): |
| """Construye el contenido del fichero ASS con el karaoke por palabra.""" |
| order = speaker_order(chunks) |
| phrases = group_into_phrases(chunks) |
|
|
| lines = [ |
| "[Script Info]", |
| "ScriptType: v4.00+", |
| f"PlayResX: {width}", |
| f"PlayResY: {height}", |
| "WrapStyle: 0", |
| "ScaledBorderAndShadow: yes", |
| "", |
| "[V4+ Styles]", |
| "Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, " |
| "OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, " |
| "ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, " |
| "MarginL, MarginR, MarginV, Encoding", |
| ] |
|
|
| |
| |
| for speaker, index in order.items(): |
| background = SPEAKER_COLORS[index % len(SPEAKER_COLORS)] |
| text_color = _text_color_for(background) |
| lines.append( |
| f"Style: S{index},{FONT_NAME},{FONT_SIZE}," |
| f"{_hex_to_ass(text_color)},{_hex_to_ass(_dim(text_color))}," |
| f"{_hex_to_ass(background)},{_hex_to_ass(background)}," |
| f"-1,0,0,0,100,100,0,0,3,6,0,2,60,60,60,1" |
| ) |
|
|
| lines += ["", "[Events]", |
| "Format: Layer, Start, End, Style, Name, MarginL, MarginR, " |
| "MarginV, Effect, Text"] |
|
|
| for phrase in phrases: |
| index = order[phrase["speaker"]] |
| parts = [] |
| cursor = phrase["start"] |
|
|
| for word in phrase["words"]: |
| |
| |
| gap = int(round((word["start"] - cursor) * 100)) |
| if gap > 0: |
| parts.append(f"{{\\k{gap}}}") |
|
|
| duration = max(1, int(round((word["end"] - word["start"]) * 100))) |
| parts.append(f"{{\\k{duration}}}{word['text']} ") |
| cursor = word["end"] |
|
|
| text = "".join(parts).strip() |
| lines.append( |
| f"Dialogue: 0,{_timestamp(phrase['start'])},{_timestamp(phrase['end'])}," |
| f"S{index},,0,0,0,,{text}" |
| ) |
|
|
| return "\n".join(lines) + "\n" |
|
|
|
|
| def get_video_size(video_path): |
| """Devuelve (ancho, alto) del vídeo, para que el ASS use su misma resolución.""" |
| result = subprocess.run( |
| ["ffprobe", "-v", "error", "-select_streams", "v:0", |
| "-show_entries", "stream=width,height", "-of", "csv=p=0:s=x", video_path], |
| capture_output=True, text=True, check=True, |
| ) |
| width, height = result.stdout.strip().split("x")[:2] |
| return int(width), int(height) |
|
|
|
|
| def render_subtitled_video(video_path, chunks, output_path, ass_path=None): |
| """Incrusta los subtítulos en el vídeo. |
| |
| Args: |
| video_path: vídeo original. |
| chunks: lista de palabras con start, end, text y speaker. |
| output_path: dónde guardar el vídeo resultante. |
| ass_path: dónde dejar el ASS generado; junto al vídeo si no se indica. |
| |
| Returns: |
| str: ruta del vídeo generado. |
| """ |
| if not chunks: |
| raise ValueError("No hay palabras que subtitular") |
|
|
| width, height = get_video_size(video_path) |
| ass_path = ass_path or os.path.splitext(output_path)[0] + ".ass" |
|
|
| os.makedirs(os.path.dirname(os.path.abspath(output_path)), exist_ok=True) |
| with open(ass_path, "w", encoding="utf-8") as f: |
| f.write(build_ass(chunks, width, height)) |
|
|
| |
| escaped = ass_path.replace("\\", "\\\\").replace(":", r"\:").replace("'", r"\'") |
|
|
| subprocess.run( |
| ["ffmpeg", "-y", "-v", "error", "-i", video_path, |
| "-vf", f"subtitles='{escaped}'", |
| "-c:a", "copy", output_path], |
| check=True, |
| ) |
|
|
| return output_path |
|
|