Spaces:
Build error
Build error
| # import numpy as np | |
| # import soundfile as sf | |
| # import torchaudio | |
| # from transformers import pipeline | |
| import os | |
| import yt_dlp | |
| import streamlit as st | |
| import torch | |
| import whisper | |
| import google.generativeai as genai | |
| from pyannote.audio import Pipeline | |
| from pyannote.audio.pipelines.utils.hook import ProgressHook | |
| from moviepy.editor import AudioFileClip | |
| from openai import Client | |
| from dotenv import load_dotenv | |
| load_dotenv() | |
| hf_token = os.environ.get("HF_DIARIZATION_TOKEN") | |
| device = torch.device("cuda" if torch.cuda.is_available() else "cpu") | |
| st.info(f"Usando dispositivo: {device}") | |
| pyannote_pipeline = Pipeline.from_pretrained( | |
| "pyannote/speaker-diarization-3.1", | |
| use_auth_token=hf_token, | |
| ) | |
| if torch.cuda.is_available(): | |
| pyannote_pipeline.to(torch.device("cuda")) | |
| whisper_model = whisper.load_model("turbo") | |
| def transcribir_segmentos(diarization_segments, audio_path): | |
| diarization_text = "" | |
| audio_duration = get_audio_duration(audio_path) | |
| path = 'diarization_transcription.txt' | |
| if os.path.exists(path): | |
| os.remove(path) | |
| st.success("Borrado archivo diarization_transcription.txt anterior") | |
| progress_bar = st.progress(0) | |
| total_segments = len(diarization_segments) | |
| with open("diarization_transcription.txt", "w", encoding="utf-8") as file: | |
| for i, (start_time, end_time, speaker) in enumerate(diarization_segments): | |
| if start_time < 0 or end_time > audio_duration: | |
| st.error(f"Segmento fuera de los límites del audio: {start_time} - {end_time}") | |
| continue | |
| if end_time - start_time < 0.5: | |
| continue | |
| segment_audio_path = f"segment_{start_time}_{end_time}.wav" | |
| extract_audio_segment(audio_path, segment_audio_path, start_time, end_time) | |
| transcript = transcribe_audio_whisper_lib(segment_audio_path) # Cambiar aca el trancript | |
| if transcript: | |
| diarization_text += f"{speaker}: {transcript}\n" | |
| file.write(f"{speaker}: {transcript}\n") | |
| os.remove(segment_audio_path) | |
| progress = (i + 1) / total_segments | |
| progress_bar.progress(progress) | |
| return diarization_text | |
| def process_transcripts(diarization_text, speakers): | |
| speaker_transcripts = {speaker: [] for speaker in speakers} | |
| for line in diarization_text.split('\n'): | |
| if line: | |
| speaker = line.split(':')[0].split()[-1] | |
| transcript = line.split(': ')[1] | |
| speaker_transcripts[speaker].append(transcript) | |
| return speaker_transcripts | |
| def summarize_speaker_transcripts(speaker_transcripts): | |
| gapi_key = os.environ.get("GEMINI_API_KEY") | |
| genai.configure(api_key=gapi_key) | |
| model = genai.GenerativeModel("gemini-1.5-flash") | |
| for speaker, transcripts in speaker_transcripts.items(): | |
| full_text = ' '.join(transcripts) | |
| max_tokens = 1024 # Ajusta según el modelo y tus necesidades | |
| if len(full_text) > max_tokens: | |
| full_text = full_text[:max_tokens] | |
| prompt=f"Por favor, resume el siguiente texto:\n\n{full_text}" | |
| response = model.generate_content(prompt) | |
| st.text_area(f"Resumen del Hablante {speaker}", response.text, height=200) | |
| def diarize_full_audio(audio_path): | |
| try: | |
| audio_duration = get_audio_duration(audio_path) | |
| diarization_segments = [] | |
| diarization_text = "" | |
| speakers = {} | |
| speaker_counter = 0 | |
| st.info(f"Procesando el audio completo de {audio_duration} segundos...") | |
| with StreamlitProgressHook() as hook: | |
| diarization = pyannote_pipeline({ | |
| 'audio': audio_path, | |
| }, hook=hook) | |
| for segment, _, speaker in diarization.itertracks(yield_label=True): | |
| if speaker not in speakers: | |
| speakers[speaker] = f"SPEAKER_{speaker_counter:02d}" | |
| speaker_counter += 1 | |
| diarization_segments.append((segment.start, segment.end, speakers[speaker])) | |
| diarization_text += f"{segment.start:.2f} - {segment.end:.2f}: {speakers[speaker]}\n" | |
| return diarization_segments, diarization_text, speakers | |
| except Exception as e: | |
| st.error(f"Error al realizar la diarización del audio entero: {e}") | |
| return None, None, 0 | |
| def extract_audio_segment(input_audio_path, output_audio_path, start_time, end_time): | |
| try: | |
| audio = AudioFileClip(input_audio_path) | |
| audio_subclip = audio.subclip(start_time, end_time) | |
| audio_subclip.write_audiofile(output_audio_path) | |
| except Exception as e: | |
| st.error(f"Error al extraer el segmento de audio: {e}") | |
| finally: | |
| if 'audio' in locals(): | |
| audio.close() | |
| if 'audio_subclip' in locals(): | |
| audio_subclip.close() | |
| def transcribe_audio_whisper_lib(audio_path): | |
| try: | |
| transcript = whisper_model.transcribe(audio_path) | |
| return transcript["text"] | |
| except Exception as e: | |
| st.error(f"Error al transcribir el audio: {e}") | |
| st.stop() | |
| def download_youtube_audio(url, progress_bar): | |
| # No incluyas la extensión '.wav' en el 'outtmpl' | |
| output_path = os.path.join('temp_audio') | |
| if os.path.exists(output_path + '.wav'): | |
| os.remove(output_path + '.wav') | |
| st.success("Borrado archivo temp_audio.wav anterior") | |
| ydl_opts = { | |
| 'format': 'bestaudio/best', | |
| 'postprocessors': [{ | |
| 'key': 'FFmpegExtractAudio', | |
| 'preferredcodec': 'wav', | |
| 'preferredquality': '192', | |
| }], | |
| 'outtmpl': output_path, | |
| 'progress_hooks': [lambda d: update_progress(d, progress_bar)], # Usa la función de progreso | |
| 'noplaylist': True, | |
| 'user-agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) ' | |
| 'Chrome/58.0.3029.110 Safari/537.3', | |
| 'cookiefile': 'cookies.txt', # Aquí incluyes el archivo de cookies | |
| } | |
| try: | |
| with yt_dlp.YoutubeDL(ydl_opts) as ydl: | |
| ydl.download([url]) | |
| progress_bar.progress(1.0) # Asegúrate de que la barra de progreso llegue al 100% cuando termine | |
| # Verifica si el archivo existe con la extensión .wav | |
| final_output_path = output_path + '.wav' | |
| st.write("Checking if file exists:", os.path.exists(final_output_path)) | |
| return final_output_path if os.path.exists(final_output_path) else None | |
| except yt_dlp.utils.DownloadError as e: | |
| st.error(f"Ocurrió un error al descargar el audio: {e}") | |
| return None | |
| except Exception as e: | |
| st.error(f"Ocurrió un error: {e}") | |
| return None | |
| def get_audio_duration(audio_path): | |
| try: | |
| audio = AudioFileClip(audio_path) | |
| return audio.duration | |
| except Exception as e: | |
| st.error(f"Error al obtener la duración del audio: {e}") | |
| return 0 | |
| def update_progress(d, progress_bar): | |
| if d['status'] == 'downloading': | |
| total_bytes = d.get('total_bytes', None) | |
| downloaded_bytes = d.get('downloaded_bytes', 0) | |
| if total_bytes: | |
| progress = downloaded_bytes / total_bytes | |
| progress_bar.progress(progress) # Actualiza la barra de progreso con el porcentaje descargado | |
| else: | |
| progress_bar.progress(0.1) # Valor predeterminado si no se conoce el tamaño total | |
| elif d['status'] == 'finished': | |
| progress_bar.progress(0.9) # 90% cuando la descarga termina | |
| elif d['status'] == 'postprocessing': | |
| progress_bar.progress(0.95) # 95% durante el procesamiento de audio | |
| class StreamlitProgressHook(ProgressHook): | |
| def __init__(self, transient: bool = False): | |
| super().__init__(transient) | |
| self.progress_text = st.empty() | |
| def __call__(self, step_name, step_artifact, file=None, total=None, completed=None): | |
| super().__call__(step_name, step_artifact, file, total, completed) | |
| if total is not None and completed is not None: | |
| progress_message = f"{step_name:<20} ━ {'━' * int(30 * (completed / total))} {completed / total:.0%}" | |
| else: | |
| progress_message = f"{step_name:<20} ━ Progress data unavailable" | |
| self.progress_text.text(progress_message) | |