Spaces:
Paused
Paused
Upload app (1) (11).py
Browse files- app (1) (11).py +836 -0
app (1) (11).py
ADDED
|
@@ -0,0 +1,836 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# -*- coding: utf-8 -*-
|
| 2 |
+
"""
|
| 3 |
+
Centralita Torá — HuggingFace Space (Gradio)
|
| 4 |
+
Leer hebreo real + español fluido con exégesis de Groq y audio optimizado
|
| 5 |
+
"""
|
| 6 |
+
import os, glob, json, tempfile, html, asyncio, subprocess, shutil, re, time
|
| 7 |
+
import urllib.request
|
| 8 |
+
import urllib.error
|
| 9 |
+
import gradio as gr
|
| 10 |
+
import edge_tts
|
| 11 |
+
try:
|
| 12 |
+
from gtts import gTTS
|
| 13 |
+
except Exception:
|
| 14 |
+
gTTS = None
|
| 15 |
+
|
| 16 |
+
from gradio_theme_fenix import fenix_theme, FENIX_CSS
|
| 17 |
+
|
| 18 |
+
# --- CONFIGURACIÓN DE VOZ Y AUDIO ---
|
| 19 |
+
VOZ_DIVINA = "es-ES-AlvaroNeural" # voz de Dios (grave, con pitch -5Hz en generar_bloque)
|
| 20 |
+
VOZ_NARRADOR = "es-MX-JorgeNeural" # voz del narrador (distinta a la de Dios)
|
| 21 |
+
|
| 22 |
+
# Perillas de entonación para Narrador en Español
|
| 23 |
+
VOZ_RATE = "-6%"
|
| 24 |
+
VOZ_PITCH = "-22Hz"
|
| 25 |
+
|
| 26 |
+
# Fondo musical
|
| 27 |
+
FONDO = "fondo.mp3"
|
| 28 |
+
FONDO_VOL = 0.18
|
| 29 |
+
|
| 30 |
+
# IA de Estudio y Refinamiento (Groq)
|
| 31 |
+
MODELO_ESTUDIA = "openai/gpt-oss-120b"
|
| 32 |
+
CARPETA = "libros"
|
| 33 |
+
GROQ_URL = "https://api.groq.com/openai/v1/chat/completions"
|
| 34 |
+
GROQ_KEY = os.environ.get("GROQ_API_KEY")
|
| 35 |
+
|
| 36 |
+
# --- SUPABASE: LECTURA DEL TANAJ ENCUADRADO ---
|
| 37 |
+
from supabase import create_client, Client
|
| 38 |
+
|
| 39 |
+
SUPABASE_URL = os.environ.get("SUPABASE_URL") or "https://bvyzbgrokexdvrhqqrla.supabase.co"
|
| 40 |
+
SUPABASE_KEY = os.environ.get("SUPABASE_KEY") # anon/public key basta para SOLO LEER
|
| 41 |
+
TABLA_TANAJ = "traducciones" # una fila por versiculo: libro, capitulo, verso, modo, es
|
| 42 |
+
MODO_A_CARGAR = "encuadrado" # el Tanaj revisado y guardado desde tanakh-translator
|
| 43 |
+
|
| 44 |
+
_supabase: "Client | None" = None
|
| 45 |
+
if SUPABASE_URL and SUPABASE_KEY:
|
| 46 |
+
try:
|
| 47 |
+
_supabase = create_client(SUPABASE_URL, SUPABASE_KEY)
|
| 48 |
+
except Exception as e:
|
| 49 |
+
print("No pude crear el cliente de Supabase:", e)
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
def _limpiar_entidades(s: str) -> str:
|
| 53 |
+
"""Deshace entidades HTML aunque vengan DOBLEMENTE escapadas
|
| 54 |
+
(p.ej. ' ' -> ' '), que es lo que rompía el hebreo en pantalla
|
| 55 |
+
mostrando literalmente ' '."""
|
| 56 |
+
prev = None
|
| 57 |
+
out = s or ""
|
| 58 |
+
for _ in range(3):
|
| 59 |
+
if out == prev:
|
| 60 |
+
break
|
| 61 |
+
prev = out
|
| 62 |
+
out = html.unescape(out)
|
| 63 |
+
return out
|
| 64 |
+
|
| 65 |
+
|
| 66 |
+
# --- TRADUCCIÓN CON SENTIDO (GROQ), VERSÍCULO POR VERSÍCULO ---
|
| 67 |
+
# Groq traduce DESDE EL HEBREO (fuente segura, siempre presente) a un español
|
| 68 |
+
# fluido y con sentido tradicional, devolviendo UNA LÍNEA POR VERSÍCULO. Así el
|
| 69 |
+
# mismo texto con sentido se muestra alineado y se usa EXACTAMENTE en el audio.
|
| 70 |
+
def _traducir_lote_groq(pares):
|
| 71 |
+
"""
|
| 72 |
+
Traduce UN LOTE pequeño de versículos (para no chocar con el límite de
|
| 73 |
+
tokens por minuto de Groq). pares: lista de (n, hebreo, es_literal).
|
| 74 |
+
Devuelve (dict {n: español_con_sentido}, estado).
|
| 75 |
+
"""
|
| 76 |
+
fallback = {n: (es or he) for (n, he, es) in pares}
|
| 77 |
+
if not pares:
|
| 78 |
+
return fallback, "sin_pares"
|
| 79 |
+
if not GROQ_KEY:
|
| 80 |
+
return fallback, "sin_key"
|
| 81 |
+
|
| 82 |
+
# Bloque de entrada numerado: el hebreo es la fuente; el literal español,
|
| 83 |
+
# si existe, va solo como ayuda.
|
| 84 |
+
lineas_in = []
|
| 85 |
+
for (n, he, es) in pares:
|
| 86 |
+
ayuda = f" (ayuda literal: {es})" if es else ""
|
| 87 |
+
lineas_in.append(f"[{n}] {he}{ayuda}")
|
| 88 |
+
entrada = "\n".join(lineas_in)
|
| 89 |
+
|
| 90 |
+
sistema = (
|
| 91 |
+
"Eres un experto en traducción bíblica del hebreo y en exégesis tradicional. "
|
| 92 |
+
"Traduces cada versículo hebreo a un español fluido, claro y natural, "
|
| 93 |
+
"con el sentido tradicional exacto, sin perder fidelidad ni profundidad. "
|
| 94 |
+
"Reglas de salida ESTRICTAS: devuelve SOLO las traducciones, una por línea, "
|
| 95 |
+
"cada línea empezando por el número entre corchetes tal cual: '[N] traducción'. "
|
| 96 |
+
"No incluyas el texto hebreo, ni títulos, ni comentarios, ni notas. "
|
| 97 |
+
"Conserva el mismo número de versículos que recibas."
|
| 98 |
+
)
|
| 99 |
+
max_tok = min(2000, max(400, len(pares) * 90))
|
| 100 |
+
cuerpo = {
|
| 101 |
+
"model": MODELO_ESTUDIA,
|
| 102 |
+
"temperature": 0.3,
|
| 103 |
+
"max_tokens": max_tok,
|
| 104 |
+
"messages": [
|
| 105 |
+
{"role": "system", "content": sistema},
|
| 106 |
+
{"role": "user", "content": f"Traduce con sentido cada versículo:\n\n{entrada}"},
|
| 107 |
+
],
|
| 108 |
+
}
|
| 109 |
+
|
| 110 |
+
try:
|
| 111 |
+
data = json.dumps(cuerpo).encode("utf-8")
|
| 112 |
+
req = urllib.request.Request(GROQ_URL, data=data, method="POST")
|
| 113 |
+
req.add_header("Content-Type", "application/json")
|
| 114 |
+
req.add_header("Authorization", f"Bearer {GROQ_KEY}")
|
| 115 |
+
req.add_header("User-Agent", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0 Safari/537.36")
|
| 116 |
+
with urllib.request.urlopen(req, timeout=120) as r:
|
| 117 |
+
d = json.load(r)
|
| 118 |
+
salida = (d["choices"][0]["message"]["content"] or "").strip()
|
| 119 |
+
except urllib.error.HTTPError as e:
|
| 120 |
+
cuerpo_error = ""
|
| 121 |
+
try:
|
| 122 |
+
cuerpo_error = e.read().decode("utf-8", errors="replace")[:300]
|
| 123 |
+
except Exception:
|
| 124 |
+
pass
|
| 125 |
+
motivo = f"error HTTP {e.code}: {cuerpo_error or e.reason}"
|
| 126 |
+
print(f"Error traduciendo con Groq: {motivo}")
|
| 127 |
+
return fallback, motivo
|
| 128 |
+
except Exception as e:
|
| 129 |
+
motivo = f"error: {type(e).__name__}: {e}"
|
| 130 |
+
print(f"Error traduciendo con Groq: {motivo}")
|
| 131 |
+
return fallback, motivo
|
| 132 |
+
|
| 133 |
+
if not salida:
|
| 134 |
+
return fallback, "respuesta vacía de Groq"
|
| 135 |
+
|
| 136 |
+
# Parseamos las líneas '[N] texto' a un diccionario por versículo.
|
| 137 |
+
res = {}
|
| 138 |
+
for linea in salida.splitlines():
|
| 139 |
+
m = re.match(r"\s*\[?(\d+)\]?[\.\)\|:\-]?\s*(.+)$", linea.strip())
|
| 140 |
+
if m:
|
| 141 |
+
res[int(m.group(1))] = m.group(2).strip()
|
| 142 |
+
|
| 143 |
+
if not res:
|
| 144 |
+
return fallback, "no se pudo interpretar la respuesta de Groq"
|
| 145 |
+
|
| 146 |
+
# Si Groq se saltó algún versículo, se completa con el fallback.
|
| 147 |
+
final = {n: res.get(n, fallback[n]) for (n, he, es) in pares}
|
| 148 |
+
faltantes = sum(1 for (n, he, es) in pares if n not in res)
|
| 149 |
+
estado = "ok" if faltantes == 0 else f"ok (Groq omitió {faltantes} versículo(s), se completó con literal)"
|
| 150 |
+
return final, estado
|
| 151 |
+
|
| 152 |
+
|
| 153 |
+
# Tamaño de lote: capítulos largos (p.ej. Génesis 1, 31 versículos) superan el
|
| 154 |
+
# límite de tokens por minuto de Groq si se mandan de una vez. Se trocea en
|
| 155 |
+
# bloques pequeños y se hacen varias llamadas, uniendo los resultados.
|
| 156 |
+
TAMANO_LOTE_GROQ = 6
|
| 157 |
+
|
| 158 |
+
|
| 159 |
+
def _segundos_de_espera(motivo_error):
|
| 160 |
+
"""Si el error es un 429 con 'Please try again in X s', devuelve X (+1s de margen)."""
|
| 161 |
+
m = re.search(r"try again in ([\d.]+)s", motivo_error)
|
| 162 |
+
if m:
|
| 163 |
+
try:
|
| 164 |
+
return float(m.group(1)) + 1.0
|
| 165 |
+
except ValueError:
|
| 166 |
+
pass
|
| 167 |
+
return None
|
| 168 |
+
|
| 169 |
+
|
| 170 |
+
def traducir_capitulo_con_groq(pares):
|
| 171 |
+
"""
|
| 172 |
+
pares: lista de (n, hebreo, es_literal) de TODO el capítulo.
|
| 173 |
+
Trocea en lotes pequeños, traduce cada uno con Groq y une los resultados.
|
| 174 |
+
Si Groq responde 429 (límite de tokens por minuto), espera el tiempo que
|
| 175 |
+
Groq indica y reintenta ese lote (hasta 3 intentos) antes de rendirse.
|
| 176 |
+
Devuelve (dict {n: español_con_sentido}, estado_general).
|
| 177 |
+
"""
|
| 178 |
+
if not pares:
|
| 179 |
+
return {}, "sin_pares"
|
| 180 |
+
|
| 181 |
+
resultado = {}
|
| 182 |
+
lotes_ok = 0
|
| 183 |
+
lotes_total = 0
|
| 184 |
+
motivos_fallo = []
|
| 185 |
+
|
| 186 |
+
for i in range(0, len(pares), TAMANO_LOTE_GROQ):
|
| 187 |
+
lote = pares[i:i + TAMANO_LOTE_GROQ]
|
| 188 |
+
lotes_total += 1
|
| 189 |
+
|
| 190 |
+
parcial, estado_lote = _traducir_lote_groq(lote)
|
| 191 |
+
intentos = 1
|
| 192 |
+
while "error HTTP 429" in estado_lote and intentos < 3:
|
| 193 |
+
espera = _segundos_de_espera(estado_lote) or 8.0
|
| 194 |
+
time.sleep(espera)
|
| 195 |
+
parcial, estado_lote = _traducir_lote_groq(lote)
|
| 196 |
+
intentos += 1
|
| 197 |
+
|
| 198 |
+
resultado.update(parcial)
|
| 199 |
+
if estado_lote == "ok" or estado_lote.startswith("ok ("):
|
| 200 |
+
lotes_ok += 1
|
| 201 |
+
else:
|
| 202 |
+
motivos_fallo.append(f"versículos {lote[0][0]}-{lote[-1][0]}: {estado_lote}")
|
| 203 |
+
|
| 204 |
+
if lotes_ok == lotes_total:
|
| 205 |
+
estado_general = "ok"
|
| 206 |
+
elif lotes_ok == 0:
|
| 207 |
+
estado_general = "; ".join(motivos_fallo[:3])
|
| 208 |
+
else:
|
| 209 |
+
estado_general = (
|
| 210 |
+
f"parcial ({lotes_ok}/{lotes_total} lotes con Groq, el resto literal) — "
|
| 211 |
+
+ "; ".join(motivos_fallo[:3])
|
| 212 |
+
)
|
| 213 |
+
|
| 214 |
+
return resultado, estado_general
|
| 215 |
+
|
| 216 |
+
|
| 217 |
+
# --- PROCESAMIENTO ACÚSTICO FFMPEG ---
|
| 218 |
+
def procesar_audio_narrador(ruta):
|
| 219 |
+
if not shutil.which("ffmpeg"):
|
| 220 |
+
return ruta
|
| 221 |
+
filtros = [
|
| 222 |
+
"atempo=0.92",
|
| 223 |
+
"bass=g=5:f=120",
|
| 224 |
+
"treble=g=2",
|
| 225 |
+
"aecho=0.8:0.88:40:0.15"
|
| 226 |
+
]
|
| 227 |
+
salida = ruta[:-4] + "_narrador.mp3"
|
| 228 |
+
cmd = ["ffmpeg", "-y", "-i", ruta, "-af", ", ".join(filtros), "-ac", "2", salida]
|
| 229 |
+
try:
|
| 230 |
+
subprocess.run(cmd, check=True, capture_output=True, timeout=60)
|
| 231 |
+
if os.path.getsize(salida) > 0:
|
| 232 |
+
return salida
|
| 233 |
+
except Exception:
|
| 234 |
+
pass
|
| 235 |
+
return ruta
|
| 236 |
+
|
| 237 |
+
|
| 238 |
+
def procesar_audio_divino(ruta):
|
| 239 |
+
if not shutil.which("ffmpeg"):
|
| 240 |
+
return ruta
|
| 241 |
+
filtros = [
|
| 242 |
+
"atempo=0.90", # Más pausado y solemne
|
| 243 |
+
"asetrate=44100*0.55", # Bajar tono (más grave y imponente)
|
| 244 |
+
"aresample=144000", # Reajustar sample rate
|
| 245 |
+
"bass=g=15:f=200", # Graves profundos sub-bass
|
| 246 |
+
"treble=g=-15", # Sonido cálido, sin aristas agudas
|
| 247 |
+
"volume=2.0", # Voz divina al doble de volumen (por encima del narrador, que queda en 1.0)
|
| 248 |
+
]
|
| 249 |
+
salida = ruta[:-4] + "_divino.mp3"
|
| 250 |
+
cmd = ["ffmpeg", "-y", "-i", ruta, "-af", ", ".join(filtros), "-ac", "2", salida]
|
| 251 |
+
try:
|
| 252 |
+
subprocess.run(cmd, check=True, capture_output=True, timeout=60)
|
| 253 |
+
if os.path.getsize(salida) > 0:
|
| 254 |
+
return salida
|
| 255 |
+
except Exception:
|
| 256 |
+
pass
|
| 257 |
+
return ruta
|
| 258 |
+
|
| 259 |
+
|
| 260 |
+
def _duracion_audio(ruta):
|
| 261 |
+
"""Duración en segundos de un archivo de audio, vía ffprobe. Se usa para
|
| 262 |
+
calcular en qué instante empieza y termina cada versículo dentro del
|
| 263 |
+
audio final, y así poder iluminar la frase que se está locutando."""
|
| 264 |
+
if not (ruta and os.path.exists(ruta) and shutil.which("ffprobe")):
|
| 265 |
+
return 0.0
|
| 266 |
+
try:
|
| 267 |
+
cmd = ["ffprobe", "-v", "error", "-show_entries", "format=duration",
|
| 268 |
+
"-of", "default=noprint_wrapped=1:nokey=1", ruta]
|
| 269 |
+
out = subprocess.run(cmd, capture_output=True, text=True, timeout=30).stdout.strip()
|
| 270 |
+
return float(out)
|
| 271 |
+
except Exception:
|
| 272 |
+
return 0.0
|
| 273 |
+
|
| 274 |
+
|
| 275 |
+
def concatenar_audios(lista_rutas):
|
| 276 |
+
if not lista_rutas:
|
| 277 |
+
return None
|
| 278 |
+
if len(lista_rutas) == 1:
|
| 279 |
+
return lista_rutas[0]
|
| 280 |
+
|
| 281 |
+
salida_final = tempfile.NamedTemporaryFile(suffix=".mp3", delete=False).name
|
| 282 |
+
list_file = tempfile.NamedTemporaryFile(suffix=".txt", mode="w", delete=False, encoding="utf-8")
|
| 283 |
+
|
| 284 |
+
for r in lista_rutas:
|
| 285 |
+
list_file.write(f"file '{os.path.abspath(r)}'\n")
|
| 286 |
+
list_file.close()
|
| 287 |
+
|
| 288 |
+
cmd = ["ffmpeg", "-y", "-f", "concat", "-safe", "0", "-i", list_file.name, "-c", "copy", salida_final]
|
| 289 |
+
try:
|
| 290 |
+
subprocess.run(cmd, check=True, capture_output=True, timeout=120)
|
| 291 |
+
os.remove(list_file.name)
|
| 292 |
+
return salida_final
|
| 293 |
+
except Exception:
|
| 294 |
+
if os.path.exists(list_file.name):
|
| 295 |
+
os.remove(list_file.name)
|
| 296 |
+
return lista_rutas[0]
|
| 297 |
+
|
| 298 |
+
|
| 299 |
+
def mezclar_fondo(voz_path):
|
| 300 |
+
if not (os.path.exists(FONDO) and shutil.which("ffmpeg")):
|
| 301 |
+
return voz_path
|
| 302 |
+
salida = voz_path[:-4] + "_mix.mp3"
|
| 303 |
+
filtro = (f"[1:a]volume={FONDO_VOL}[m];"
|
| 304 |
+
f"[0:a][m]amix=inputs=2:duration=first:normalize=0[a]")
|
| 305 |
+
cmd = ["ffmpeg", "-y", "-i", voz_path, "-stream_loop", "-1", "-i", FONDO,
|
| 306 |
+
"-filter_complex", filtro, "-map", "[a]", salida]
|
| 307 |
+
try:
|
| 308 |
+
subprocess.run(cmd, check=True, capture_output=True, timeout=120)
|
| 309 |
+
if os.path.getsize(salida) > 0:
|
| 310 |
+
return salida
|
| 311 |
+
except Exception:
|
| 312 |
+
pass
|
| 313 |
+
return voz_path
|
| 314 |
+
|
| 315 |
+
|
| 316 |
+
# --- SEPARACIÓN DE BLOQUES (NARRADOR vs DIOS) PARA ESPAÑOL ---
|
| 317 |
+
# Fórmulas de NARRACIÓN que, si aparecen dentro de lo que dice Dios, marcan
|
| 318 |
+
# que la voz divina ya terminó y vuelve el narrador (ej. "y fue así").
|
| 319 |
+
_FIN_DIOS = re.compile(
|
| 320 |
+
r'\s*(?:;?\s*y\s+(?:fue\s+as[ií]|as[ií]\s+fue|existir\s+poner\s+erguido|'
|
| 321 |
+
r'fue\s+la\s+(?:luz|tarde|ma[nñ]ana)|vio\s+Dios|ver\s+Dios).*)$',
|
| 322 |
+
flags=re.IGNORECASE)
|
| 323 |
+
|
| 324 |
+
|
| 325 |
+
def segmentar_texto_divino(texto):
|
| 326 |
+
# Verbos de "hablar" que introducen palabras de Dios (formas del literal
|
| 327 |
+
# incluidas: decir/dijo, llamar/llamó, bendecir/bendijo, ordenar, mandar...).
|
| 328 |
+
verbo = (r'(?:dijo|dice|decir|diciendo|para\s+decir|llam[óo]|llamar|'
|
| 329 |
+
r'bendijo|bendecir|habl[óo]|hablar|orden[óo]|mand[óo]|respondi[óo])')
|
| 330 |
+
patron = re.compile(
|
| 331 |
+
r'((?:y\s+)?' + verbo + r'\s+Dios|Dios\s+' + verbo + r')' # intro (narrador)
|
| 332 |
+
r'([:,.\-–—]?\s*)' # separador
|
| 333 |
+
r'([^.\n]+)', # lo que dice Dios
|
| 334 |
+
flags=re.IGNORECASE)
|
| 335 |
+
|
| 336 |
+
bloques = []
|
| 337 |
+
ultimo = 0
|
| 338 |
+
for m in patron.finditer(texto):
|
| 339 |
+
ini, fin = m.span()
|
| 340 |
+
if ini > ultimo:
|
| 341 |
+
bloques.append(("narrador", texto[ultimo:ini]))
|
| 342 |
+
|
| 343 |
+
intro = m.group(1) + (m.group(2) or "")
|
| 344 |
+
dicho = m.group(3)
|
| 345 |
+
|
| 346 |
+
# Si dentro de lo dicho aparece una fórmula de narración, la separamos.
|
| 347 |
+
corte = _FIN_DIOS.search(dicho)
|
| 348 |
+
if corte:
|
| 349 |
+
palabras_dios = dicho[:corte.start()]
|
| 350 |
+
cola_narrador = dicho[corte.start():]
|
| 351 |
+
else:
|
| 352 |
+
palabras_dios, cola_narrador = dicho, ""
|
| 353 |
+
|
| 354 |
+
bloques.append(("narrador", intro))
|
| 355 |
+
if palabras_dios.strip():
|
| 356 |
+
bloques.append(("dios", palabras_dios))
|
| 357 |
+
if cola_narrador.strip():
|
| 358 |
+
bloques.append(("narrador", cola_narrador))
|
| 359 |
+
ultimo = fin
|
| 360 |
+
|
| 361 |
+
if ultimo < len(texto):
|
| 362 |
+
bloques.append(("narrador", texto[ultimo:]))
|
| 363 |
+
|
| 364 |
+
return bloques if bloques else [("narrador", texto)]
|
| 365 |
+
|
| 366 |
+
|
| 367 |
+
async def generar_bloque_audio(texto, es_divino):
|
| 368 |
+
ruta = tempfile.NamedTemporaryFile(suffix=".mp3", delete=False).name
|
| 369 |
+
voz = VOZ_DIVINA if es_divino else VOZ_NARRADOR
|
| 370 |
+
pitch = "-5Hz" if es_divino else VOZ_PITCH
|
| 371 |
+
rate = "-9%" if es_divino else VOZ_RATE
|
| 372 |
+
|
| 373 |
+
try:
|
| 374 |
+
com = edge_tts.Communicate(texto[:4000], voz, rate=rate, pitch=pitch)
|
| 375 |
+
await com.save(ruta)
|
| 376 |
+
except Exception:
|
| 377 |
+
return None
|
| 378 |
+
|
| 379 |
+
if es_divino:
|
| 380 |
+
return procesar_audio_divino(ruta)
|
| 381 |
+
else:
|
| 382 |
+
return procesar_audio_narrador(ruta)
|
| 383 |
+
|
| 384 |
+
|
| 385 |
+
async def leer_es(pares_es):
|
| 386 |
+
"""pares_es: lista de (n_verso, texto_es) a locutar en orden.
|
| 387 |
+
Genera el audio versículo por versículo (cada uno puede a su vez partirse
|
| 388 |
+
en narrador/dios) y devuelve, además del audio final, un timeline JSON
|
| 389 |
+
con el inicio y fin (en segundos) de cada versículo dentro del audio,
|
| 390 |
+
para poder iluminar en pantalla la frase que se está reproduciendo."""
|
| 391 |
+
if not pares_es:
|
| 392 |
+
return None, "[]"
|
| 393 |
+
|
| 394 |
+
audios_segmentos = []
|
| 395 |
+
duracion_por_verso = {} # n_verso -> segundos acumulados de sus fragmentos
|
| 396 |
+
|
| 397 |
+
for n_verso, texto in pares_es:
|
| 398 |
+
texto = (texto or "").strip()
|
| 399 |
+
if not texto:
|
| 400 |
+
continue
|
| 401 |
+
bloques = segmentar_texto_divino(texto)
|
| 402 |
+
for tipo, contenido in bloques:
|
| 403 |
+
if not contenido.strip():
|
| 404 |
+
continue
|
| 405 |
+
es_divino = (tipo == "dios")
|
| 406 |
+
audio_seg = await generar_bloque_audio(contenido, es_divino)
|
| 407 |
+
if audio_seg:
|
| 408 |
+
audios_segmentos.append(audio_seg)
|
| 409 |
+
duracion_por_verso[n_verso] = duracion_por_verso.get(n_verso, 0.0) + _duracion_audio(audio_seg)
|
| 410 |
+
|
| 411 |
+
audio_final = concatenar_audios(audios_segmentos)
|
| 412 |
+
|
| 413 |
+
# El timeline se calcula ANTES de mezclar el fondo musical (mezclar_fondo
|
| 414 |
+
# conserva la duración total, duration=first, así que los tiempos siguen
|
| 415 |
+
# siendo válidos igual).
|
| 416 |
+
timeline = []
|
| 417 |
+
t = 0.0
|
| 418 |
+
for n_verso, texto in pares_es:
|
| 419 |
+
dur = duracion_por_verso.get(n_verso, 0.0)
|
| 420 |
+
if dur <= 0:
|
| 421 |
+
continue
|
| 422 |
+
timeline.append({"verso": n_verso, "start": round(t, 3), "end": round(t + dur, 3)})
|
| 423 |
+
t += dur
|
| 424 |
+
|
| 425 |
+
if audio_final:
|
| 426 |
+
audio_final = mezclar_fondo(audio_final)
|
| 427 |
+
|
| 428 |
+
return audio_final, json.dumps(timeline, ensure_ascii=False)
|
| 429 |
+
|
| 430 |
+
|
| 431 |
+
# --- REPRODUCTOR DE HEBREO LITÚRGICO REAL (TROPE) ---
|
| 432 |
+
async def leer_he(nombre_espanol, capitulo):
|
| 433 |
+
libro_id = id_por_nombre(nombre_espanol)
|
| 434 |
+
if not libro_id:
|
| 435 |
+
return None
|
| 436 |
+
|
| 437 |
+
MAPEO_TORAH = {
|
| 438 |
+
"Genesis": "01",
|
| 439 |
+
"Exodus": "02",
|
| 440 |
+
"Leviticus": "03",
|
| 441 |
+
"Numbers": "04",
|
| 442 |
+
"Deuteronomy": "05"
|
| 443 |
+
}
|
| 444 |
+
|
| 445 |
+
carpeta_audios = "audios_torah"
|
| 446 |
+
if not os.path.exists(carpeta_audios):
|
| 447 |
+
os.makedirs(carpeta_audios)
|
| 448 |
+
|
| 449 |
+
ruta_local = os.path.join(carpeta_audios, f"{libro_id}_{capitulo}.mp3")
|
| 450 |
+
|
| 451 |
+
if os.path.exists(ruta_local):
|
| 452 |
+
return mezclar_fondo(ruta_local)
|
| 453 |
+
|
| 454 |
+
if libro_id in MAPEO_TORAH:
|
| 455 |
+
id_mechon = MAPEO_TORAH[libro_id]
|
| 456 |
+
cap_formateado = f"{int(capitulo):02d}"
|
| 457 |
+
url_audio = f"https://mechon-mamre.org/mp3/t{id_mechon}{cap_formateado}.mp3"
|
| 458 |
+
|
| 459 |
+
try:
|
| 460 |
+
req = urllib.request.Request(
|
| 461 |
+
url_audio,
|
| 462 |
+
headers={"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
|
| 463 |
+
)
|
| 464 |
+
with urllib.request.urlopen(req, timeout=30) as response, open(ruta_local, 'wb') as out_file:
|
| 465 |
+
out_file.write(response.read())
|
| 466 |
+
return mezclar_fondo(ruta_local)
|
| 467 |
+
except Exception as e:
|
| 468 |
+
print(f"Error descargando el audio: {e}")
|
| 469 |
+
return None
|
| 470 |
+
|
| 471 |
+
return None
|
| 472 |
+
|
| 473 |
+
|
| 474 |
+
# --- CARGA Y MANEJO DE LIBROS ---
|
| 475 |
+
NOMBRES_ES = {
|
| 476 |
+
"Genesis": "Génesis", "Exodus": "Éxodo", "Leviticus": "Levítico",
|
| 477 |
+
"Numbers": "Números", "Deuteronomy": "Deuteronomio", "Joshua": "Josué",
|
| 478 |
+
"Judges": "Jueces", "Ruth": "Rut", "1_Samuel": "1 Samuel", "2_Samuel": "2 Samuel",
|
| 479 |
+
"1_Kings": "1 Reyes", "2_Kings": "2 Reyes", "1_Chronicles": "1 Crónicas",
|
| 480 |
+
"2_Chronicles": "2 Crónicas", "Ezra": "Esdras", "Nehemiah": "Nehemías",
|
| 481 |
+
"Esther": "Ester", "Job": "Job", "Psalms": "Salmos", "Proverbs": "Proverbios",
|
| 482 |
+
"Ecclesiastes": "Eclesiastés", "Song_of_Songs": "Cantar de los Cantares",
|
| 483 |
+
"Isaiah": "Isaías", "Jeremiah": "Jeremías", "Lamentations": "Lamentaciones",
|
| 484 |
+
"Ezekiel": "Ezequiel", "Daniel": "Daniel", "Hosea": "Oseas", "Joel": "Joel",
|
| 485 |
+
"Amos": "Amós", "Obadiah": "Abdías", "Jonah": "Jonás", "Micah": "Miqueas",
|
| 486 |
+
"Nahum": "Nahúm", "Habakkuk": "Habacuc", "Zephaniah": "Sofonías",
|
| 487 |
+
"Haggai": "Hageo", "Zechariah": "Zacarías", "Malachi": "Malaquías",
|
| 488 |
+
}
|
| 489 |
+
|
| 490 |
+
|
| 491 |
+
def _convertir_traduccion_plana(book_id: str, data: dict) -> dict:
|
| 492 |
+
texto = data.get("translation", "") or ""
|
| 493 |
+
lineas = [l.strip() for l in texto.split("\n") if l.strip()]
|
| 494 |
+
versiculos = [{"n": i + 1, "he": "", "es": linea} for i, linea in enumerate(lineas)]
|
| 495 |
+
return {
|
| 496 |
+
"id": book_id,
|
| 497 |
+
"es": NOMBRES_ES.get(book_id, book_id),
|
| 498 |
+
"heb": "",
|
| 499 |
+
"capitulos": {"1": versiculos},
|
| 500 |
+
}
|
| 501 |
+
|
| 502 |
+
|
| 503 |
+
def cargar_libros():
|
| 504 |
+
"""Carga los libros del Tanaj desde Supabase (tabla TABLA_TANAJ, modo=MODO_A_CARGAR,
|
| 505 |
+
actualmente 'encuadrado' = el texto revisado y guardado desde tanakh-translator).
|
| 506 |
+
Cada fila esperada = un versículo, con columnas: libro, capitulo, verso, es."""
|
| 507 |
+
libros = {}
|
| 508 |
+
|
| 509 |
+
if not _supabase:
|
| 510 |
+
print("Faltan los Secrets SUPABASE_URL / SUPABASE_KEY en el Space "
|
| 511 |
+
"(Settings → Variables and secrets). Cargando 0 libros.")
|
| 512 |
+
return _cargar_libros_locales_fallback()
|
| 513 |
+
|
| 514 |
+
# La tabla 'traducciones' guarda UNA FILA POR VERSICULO. Traemos todas las
|
| 515 |
+
# del modo pedido, paginando (Supabase corta a 1000 por defecto), y las
|
| 516 |
+
# agrupamos por libro y capitulo. El hebreo se lee de la carpeta libros/.
|
| 517 |
+
filas = []
|
| 518 |
+
desde = 0
|
| 519 |
+
PASO = 1000
|
| 520 |
+
try:
|
| 521 |
+
while True:
|
| 522 |
+
resp = (
|
| 523 |
+
_supabase.table(TABLA_TANAJ)
|
| 524 |
+
.select("libro,capitulo,verso,es")
|
| 525 |
+
.eq("modo", MODO_A_CARGAR)
|
| 526 |
+
.range(desde, desde + PASO - 1)
|
| 527 |
+
.execute()
|
| 528 |
+
)
|
| 529 |
+
lote = resp.data or []
|
| 530 |
+
filas.extend(lote)
|
| 531 |
+
if len(lote) < PASO:
|
| 532 |
+
break
|
| 533 |
+
desde += PASO
|
| 534 |
+
except Exception as e:
|
| 535 |
+
print(f"No pude leer la tabla '{TABLA_TANAJ}' en Supabase: {e}")
|
| 536 |
+
return _cargar_libros_locales_fallback()
|
| 537 |
+
|
| 538 |
+
# Hebreo por versiculo desde libros/ (para mostrarlo arriba).
|
| 539 |
+
heb_por = {}
|
| 540 |
+
for ruta in glob.glob(os.path.join(CARPETA, "*.json")):
|
| 541 |
+
try:
|
| 542 |
+
with open(ruta, encoding="utf-8") as f:
|
| 543 |
+
d = json.load(f)
|
| 544 |
+
lid = d.get("id") or os.path.splitext(os.path.basename(ruta))[0]
|
| 545 |
+
for cap, versos in d.get("capitulos", {}).items():
|
| 546 |
+
for i, v in enumerate(versos):
|
| 547 |
+
n = v.get("n", i + 1)
|
| 548 |
+
heb_por[(lid, str(cap), n)] = v.get("he", "")
|
| 549 |
+
except Exception:
|
| 550 |
+
pass
|
| 551 |
+
|
| 552 |
+
# Agrupar versiculos por libro -> capitulo -> lista ordenada.
|
| 553 |
+
tmp = {}
|
| 554 |
+
for fila in filas:
|
| 555 |
+
lid = fila.get("libro")
|
| 556 |
+
cap = str(fila.get("capitulo"))
|
| 557 |
+
n = fila.get("verso")
|
| 558 |
+
if lid is None or n is None:
|
| 559 |
+
continue
|
| 560 |
+
tmp.setdefault(lid, {}).setdefault(cap, []).append({
|
| 561 |
+
"n": n,
|
| 562 |
+
"es": fila.get("es", ""),
|
| 563 |
+
"he": heb_por.get((lid, cap, n), ""),
|
| 564 |
+
})
|
| 565 |
+
|
| 566 |
+
for lid, caps in tmp.items():
|
| 567 |
+
capitulos = {}
|
| 568 |
+
for cap, versos in caps.items():
|
| 569 |
+
capitulos[cap] = sorted(versos, key=lambda v: v["n"])
|
| 570 |
+
libros[lid] = {
|
| 571 |
+
"id": lid,
|
| 572 |
+
"es": NOMBRES_ES.get(lid, lid),
|
| 573 |
+
"heb": "",
|
| 574 |
+
"capitulos": capitulos,
|
| 575 |
+
}
|
| 576 |
+
|
| 577 |
+
if not libros:
|
| 578 |
+
print(f"Supabase respondio pero no llego ningun versiculo con modo='{MODO_A_CARGAR}' "
|
| 579 |
+
f"en la tabla '{TABLA_TANAJ}'. Revisa nombre de tabla/columnas.")
|
| 580 |
+
|
| 581 |
+
return libros
|
| 582 |
+
|
| 583 |
+
|
| 584 |
+
def _cargar_libros_locales_fallback():
|
| 585 |
+
"""Si Supabase no está configurado o falla, intenta la carpeta local libros/ como respaldo."""
|
| 586 |
+
libros = {}
|
| 587 |
+
for ruta in sorted(glob.glob(os.path.join(CARPETA, "*.json"))):
|
| 588 |
+
try:
|
| 589 |
+
with open(ruta, encoding="utf-8") as f:
|
| 590 |
+
d = json.load(f)
|
| 591 |
+
book_id = os.path.splitext(os.path.basename(ruta))[0]
|
| 592 |
+
if "capitulos" in d:
|
| 593 |
+
lid = d.get("id") or book_id
|
| 594 |
+
libros[lid] = d
|
| 595 |
+
elif "translation" in d:
|
| 596 |
+
libros[book_id] = _convertir_traduccion_plana(book_id, d)
|
| 597 |
+
except Exception as e:
|
| 598 |
+
print("No pude leer", ruta, e)
|
| 599 |
+
return libros
|
| 600 |
+
|
| 601 |
+
|
| 602 |
+
LIBROS = cargar_libros()
|
| 603 |
+
ORDEN = ["Genesis", "Exodus", "Leviticus", "Numbers", "Deuteronomy"]
|
| 604 |
+
IDS = [i for i in ORDEN if i in LIBROS] + [i for i in LIBROS if i not in ORDEN]
|
| 605 |
+
NOMBRES = [LIBROS[i].get("es", i) for i in IDS] or ["(sube tus libros)"]
|
| 606 |
+
|
| 607 |
+
|
| 608 |
+
def id_por_nombre(nombre):
|
| 609 |
+
for i in IDS:
|
| 610 |
+
if LIBROS[i].get("es", i) == nombre:
|
| 611 |
+
return i
|
| 612 |
+
return IDS[0] if IDS else None
|
| 613 |
+
|
| 614 |
+
|
| 615 |
+
def versiculos(lid, cap):
|
| 616 |
+
return LIBROS.get(lid, {}).get("capitulos", {}).get(str(cap), [])
|
| 617 |
+
|
| 618 |
+
|
| 619 |
+
def render(nombre, cap):
|
| 620 |
+
lid = id_por_nombre(nombre)
|
| 621 |
+
if not lid:
|
| 622 |
+
return "<div class='pasaje'><p>Aún no has subido libros a la carpeta <b>libros/</b>.</p></div>", "", ""
|
| 623 |
+
try:
|
| 624 |
+
cap = max(1, int(float(cap or 1)))
|
| 625 |
+
except Exception:
|
| 626 |
+
cap = 1
|
| 627 |
+
d = LIBROS[lid]
|
| 628 |
+
vs = versiculos(lid, cap)
|
| 629 |
+
if not vs:
|
| 630 |
+
return f"<div class='pasaje'><p>No hay texto para {html.escape(d.get('es',''))} {cap}.</p></div>", "", ""
|
| 631 |
+
|
| 632 |
+
# Recopilamos (n, hebreo_limpio, español_literal) por versículo.
|
| 633 |
+
pares = []
|
| 634 |
+
plano_he = []
|
| 635 |
+
for i, v in enumerate(vs):
|
| 636 |
+
n_verso = v.get('n', i + 1)
|
| 637 |
+
he_raw = _limpiar_entidades(v.get("he", "")).replace(" ", " ").replace("{פ}", "").replace("{ס}", "").strip()
|
| 638 |
+
es_raw = _limpiar_entidades(v.get("es", "")).replace(" ", " ").strip()
|
| 639 |
+
pares.append((n_verso, he_raw, es_raw))
|
| 640 |
+
if he_raw:
|
| 641 |
+
plano_he.append(he_raw)
|
| 642 |
+
|
| 643 |
+
# El español YA viene encuadrado desde Supabase (modo='encuadrado').
|
| 644 |
+
# No se vuelve a traducir: se muestra tal cual se guardó.
|
| 645 |
+
|
| 646 |
+
# Interfaz: hebreo intacto arriba y, debajo, la traducción con sentido
|
| 647 |
+
# (sin aviso visible; el estado queda solo en los logs del Space).
|
| 648 |
+
filas = [f"<div class='cab'><span class='h'>{html.escape(_limpiar_entidades(d.get('heb','')))}</span><br>"
|
| 649 |
+
f"{html.escape(d.get('es',''))} {cap}</div>"]
|
| 650 |
+
|
| 651 |
+
partes_es = []
|
| 652 |
+
pares_es = [] # [(n_verso, texto_es)] en orden, para locutar y sincronizar el resaltado
|
| 653 |
+
for (n_verso, he_raw, es_raw) in pares:
|
| 654 |
+
es_sentido = (es_raw or "").strip()
|
| 655 |
+
if es_sentido:
|
| 656 |
+
partes_es.append(es_sentido)
|
| 657 |
+
pares_es.append((n_verso, es_sentido))
|
| 658 |
+
|
| 659 |
+
fila = f"<div class='verso' data-n='{n_verso}'><span class='num'>{n_verso}</span>"
|
| 660 |
+
if he_raw:
|
| 661 |
+
fila += f"<span class='he'>{html.escape(he_raw)}</span>"
|
| 662 |
+
if es_sentido:
|
| 663 |
+
fila += f"<span class='es'>{html.escape(es_sentido)}</span>"
|
| 664 |
+
fila += "</div>"
|
| 665 |
+
filas.append(fila)
|
| 666 |
+
|
| 667 |
+
# Texto que se locuta = exactamente la traducción con sentido de todo el capítulo.
|
| 668 |
+
texto_es_final = " ".join(partes_es)
|
| 669 |
+
|
| 670 |
+
return f"<div class='pasaje'>{''.join(filas)}</div>", texto_es_final, " ".join(plano_he), pares_es
|
| 671 |
+
|
| 672 |
+
|
| 673 |
+
def estudiar(mensaje, historial, contexto):
|
| 674 |
+
if not GROQ_KEY:
|
| 675 |
+
return "Para el estudio, añade el Secret GROQ_API_KEY en el Space."
|
| 676 |
+
sistema = (
|
| 677 |
+
"Eres un compañero de estudio de la Torá que responde en español, cálido y honesto. "
|
| 678 |
+
"Ofreces el sentido literal (peshat), contexto histórico y lingüístico, capas de la "
|
| 679 |
+
"tradición (midrash, Rashi cuando venga al caso) y reflexión espiritual. Vas al grano.\n\n" + contexto
|
| 680 |
+
)
|
| 681 |
+
mensajes = [{"role": "system", "content": sistema}]
|
| 682 |
+
for m in (historial or []):
|
| 683 |
+
if isinstance(m, dict) and m.get("role") in ("user", "assistant"):
|
| 684 |
+
mensajes.append({"role": m["role"], "content": m["content"]})
|
| 685 |
+
mensajes.append({"role": "user", "content": mensaje})
|
| 686 |
+
cuerpo = {"model": MODELO_ESTUDIA, "temperature": 0.4, "max_tokens": 1200, "messages": mensajes}
|
| 687 |
+
try:
|
| 688 |
+
data = json.dumps(cuerpo).encode("utf-8")
|
| 689 |
+
req = urllib.request.Request(GROQ_URL, data=data, method="POST")
|
| 690 |
+
req.add_header("Content-Type", "application/json")
|
| 691 |
+
req.add_header("Authorization", f"Bearer {GROQ_KEY}")
|
| 692 |
+
req.add_header("User-Agent", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0 Safari/537.36")
|
| 693 |
+
with urllib.request.urlopen(req, timeout=90) as r:
|
| 694 |
+
d = json.load(r)
|
| 695 |
+
return (d["choices"][0]["message"]["content"] or "").strip()
|
| 696 |
+
except Exception as e:
|
| 697 |
+
return f"No pude consultar a la IA ahora mismo ({e})."
|
| 698 |
+
|
| 699 |
+
|
| 700 |
+
# ============================ INTERFAZ GRADIO ============================
|
| 701 |
+
with gr.Blocks(theme=fenix_theme(), css=FENIX_CSS, title="Centralita Torá") as demo:
|
| 702 |
+
st_es = gr.State("")
|
| 703 |
+
st_he = gr.State("")
|
| 704 |
+
st_pares_es = gr.State([])
|
| 705 |
+
st_nombre = gr.State(NOMBRES[0])
|
| 706 |
+
st_cap = gr.State(1)
|
| 707 |
+
timing_box = gr.Textbox(value="[]", visible=False, elem_id="timing_data")
|
| 708 |
+
|
| 709 |
+
gr.HTML("<div id='cabecera'><div class='estrella'>✡</div>"
|
| 710 |
+
"<h1>Centralita Torá</h1><p>hebreo real e impecable · español fluido y natural</p></div>")
|
| 711 |
+
|
| 712 |
+
# Resaltado dorado, clarito y transparente, de la frase que se está locutando ahora mismo.
|
| 713 |
+
gr.HTML("""
|
| 714 |
+
<style>
|
| 715 |
+
.verso .es {
|
| 716 |
+
transition: background-color .35s ease, box-shadow .35s ease, color .35s ease;
|
| 717 |
+
border-radius: 6px;
|
| 718 |
+
padding: 1px 5px;
|
| 719 |
+
}
|
| 720 |
+
.verso.activo .es {
|
| 721 |
+
background-color: rgba(255, 213, 122, 0.25);
|
| 722 |
+
box-shadow: 0 0 14px rgba(255, 213, 122, 0.35);
|
| 723 |
+
color: #fff6dc;
|
| 724 |
+
}
|
| 725 |
+
</style>
|
| 726 |
+
""")
|
| 727 |
+
|
| 728 |
+
with gr.Tab("Leer"):
|
| 729 |
+
with gr.Row():
|
| 730 |
+
dd = gr.Dropdown(NOMBRES, value=NOMBRES[0], label="Libro")
|
| 731 |
+
ncap = gr.Number(value=1, precision=0, label="Capítulo", minimum=1)
|
| 732 |
+
ver_btn = gr.Button("Ver capítulo", variant="primary")
|
| 733 |
+
salida = gr.HTML()
|
| 734 |
+
with gr.Row():
|
| 735 |
+
b_es = gr.Button("🔊 Narrador + Voz Divina (Español)", variant="secondary")
|
| 736 |
+
b_he = gr.Button("🔊 עברית (Hebreo con Trope Real)", variant="secondary")
|
| 737 |
+
audio = gr.Audio(label="Audio Litúrgico", autoplay=True, elem_id="tts_audio")
|
| 738 |
+
|
| 739 |
+
def ver(nombre, cap):
|
| 740 |
+
md, es, he, pares_es = render(nombre, cap)
|
| 741 |
+
return md, es, he, pares_es, nombre, cap
|
| 742 |
+
|
| 743 |
+
ver_btn.click(ver, [dd, ncap], [salida, st_es, st_he, st_pares_es, st_nombre, st_cap])
|
| 744 |
+
b_es.click(leer_es, st_pares_es, [audio, timing_box])
|
| 745 |
+
b_he.click(leer_he, inputs=[st_nombre, st_cap], outputs=[audio])
|
| 746 |
+
|
| 747 |
+
with gr.Tab("Buscar"):
|
| 748 |
+
q = gr.Textbox(label="Palabra o frase (hebreo o español)", placeholder="ej. luz / אור")
|
| 749 |
+
q_btn = gr.Button("Buscar", variant="primary")
|
| 750 |
+
q_out = gr.HTML()
|
| 751 |
+
|
| 752 |
+
def buscar(texto):
|
| 753 |
+
t = (texto or "").strip().lower()
|
| 754 |
+
if not t:
|
| 755 |
+
return "<p>Escribe algo para buscar.</p>"
|
| 756 |
+
filas = []
|
| 757 |
+
for lid in IDS:
|
| 758 |
+
d = LIBROS[lid]
|
| 759 |
+
for c, vs in d.get("capitulos", {}).items():
|
| 760 |
+
for v in vs:
|
| 761 |
+
campos = " ".join(filter(None, [v.get("es"), v.get("he")])).lower()
|
| 762 |
+
if t in campos:
|
| 763 |
+
ref = f"{d.get('es', lid)} {c}:{v.get('n')}"
|
| 764 |
+
txt = html.escape(_limpiar_entidades(v.get("es") or v.get("he") or ""))
|
| 765 |
+
filas.append(f"<div class='resultado'><span class='ref'>{ref}</span> — {txt}</div>")
|
| 766 |
+
if len(filas) >= 120:
|
| 767 |
+
return "".join(filas)
|
| 768 |
+
return "".join(filas) if filas else "<p>Sin resultados en los libros cargados.</p>"
|
| 769 |
+
|
| 770 |
+
q_btn.click(buscar, q, q_out)
|
| 771 |
+
|
| 772 |
+
with gr.Tab("Estudiar"):
|
| 773 |
+
gr.Markdown("La IA usa el capítulo abierto en **Leer** como contexto.")
|
| 774 |
+
chat = gr.Chatbot(type="messages", height=360)
|
| 775 |
+
pin = gr.Textbox(placeholder="Pregunta sobre el pasaje…", label="")
|
| 776 |
+
with gr.Row():
|
| 777 |
+
enviar = gr.Button("Preguntar", variant="primary")
|
| 778 |
+
comentar = gr.Button("Comentar este capítulo", variant="secondary")
|
| 779 |
+
|
| 780 |
+
def responder(mensaje, historial, es, he, nombre, cap):
|
| 781 |
+
mensaje = (mensaje.strip() if mensaje else "")
|
| 782 |
+
if not mensaje:
|
| 783 |
+
return historial, ""
|
| 784 |
+
contexto = f"Pasaje en pantalla — {nombre} {cap}:\nHebreo: {he}\nEspañol: {es}"
|
| 785 |
+
r = estudiar(mensaje, historial, contexto)
|
| 786 |
+
historial = (historial or []) + [
|
| 787 |
+
{"role": "user", "content": mensaje},
|
| 788 |
+
{"role": "assistant", "content": r},
|
| 789 |
+
]
|
| 790 |
+
return historial, ""
|
| 791 |
+
|
| 792 |
+
enviar.click(responder, [pin, chat, st_es, st_he, st_nombre, st_cap], [chat, pin])
|
| 793 |
+
comentar.click(
|
| 794 |
+
lambda h, es, he, n, c: responder("Comenta y ayúdame a estudiar este capítulo.", h, es, he, n, c),
|
| 795 |
+
[chat, st_es, st_he, st_nombre, st_cap], [chat, pin],
|
| 796 |
+
)
|
| 797 |
+
|
| 798 |
+
with gr.Tab("Cómo subir"):
|
| 799 |
+
gr.Markdown(
|
| 800 |
+
"Sube un JSON por libro a la carpeta `libros/`. Para el estudio y traducción fluida, añade el Secret `GROQ_API_KEY`."
|
| 801 |
+
)
|
| 802 |
+
|
| 803 |
+
demo.load(None, None, None, js="""
|
| 804 |
+
() => {
|
| 805 |
+
if (window.__fenixHighlightInterval) return;
|
| 806 |
+
window.__fenixHighlightInterval = setInterval(() => {
|
| 807 |
+
const audioEl = document.querySelector('#tts_audio audio');
|
| 808 |
+
const timingBox = document.querySelector('#timing_data textarea');
|
| 809 |
+
if (!audioEl || !timingBox) return;
|
| 810 |
+
|
| 811 |
+
let timeline;
|
| 812 |
+
try { timeline = JSON.parse(timingBox.value || '[]'); } catch (e) { return; }
|
| 813 |
+
if (!Array.isArray(timeline) || timeline.length === 0) {
|
| 814 |
+
document.querySelectorAll('.verso.activo').forEach(el => el.classList.remove('activo'));
|
| 815 |
+
return;
|
| 816 |
+
}
|
| 817 |
+
|
| 818 |
+
const t = audioEl.currentTime;
|
| 819 |
+
let activo = null;
|
| 820 |
+
for (const seg of timeline) {
|
| 821 |
+
if (t >= seg.start && t < seg.end) { activo = seg.verso; break; }
|
| 822 |
+
}
|
| 823 |
+
|
| 824 |
+
document.querySelectorAll('.verso.activo').forEach(el => {
|
| 825 |
+
if (String(el.dataset.n) !== String(activo)) el.classList.remove('activo');
|
| 826 |
+
});
|
| 827 |
+
if (activo !== null) {
|
| 828 |
+
const el = document.querySelector(`.verso[data-n='${activo}']`);
|
| 829 |
+
if (el && !el.classList.contains('activo')) el.classList.add('activo');
|
| 830 |
+
}
|
| 831 |
+
}, 120);
|
| 832 |
+
}
|
| 833 |
+
""")
|
| 834 |
+
|
| 835 |
+
if __name__ == "__main__":
|
| 836 |
+
demo.launch(ssr_mode=False)
|