cap / app.py
rahulxsantra's picture
Update app.py
c38d1bb verified
Raw
History Blame Contribute Delete
4 kB
import gradio as gr
import tempfile
import os
import subprocess
import json
from openai import OpenAI
from moviepy.editor import VideoFileClip
def extract_audio(video_path):
"""Extract audio from video file and save it as a temporary mp3 file."""
temp_audio = tempfile.NamedTemporaryFile(suffix='.mp3', delete=False)
temp_audio.close()
video = VideoFileClip(video_path)
video.audio.write_audiofile(temp_audio.name, verbose=False, logger=None)
return temp_audio.name
def transcribe_with_timestamps(audio_path, api_key):
"""Transcribe audio with word-level timestamps using OpenAI's Whisper API."""
client = OpenAI(api_key=api_key)
try:
with open(audio_path, "rb") as audio_file:
response = client.audio.transcriptions.create(
model="whisper-1", # or "gpt-4o-transcribe" if available
file=audio_file,
response_format="verbose_json",
timestamp_granularities=["word"]
)
# Format the results
words = []
for segment in response.segments:
for word in segment.words:
words.append({
"word": word.word,
"start": word.start,
"end": word.end
})
return words, response.text
except Exception as e:
return None, f"Error during transcription: {str(e)}"
def process_video(video, api_key):
"""Process uploaded video, extract audio, and transcribe with timestamps."""
if not video or not api_key:
return "Please provide both a video file and an OpenAI API key.", None
try:
# Extract audio from video
audio_path = extract_audio(video)
# Get transcription with timestamps
word_timestamps, full_text = transcribe_with_timestamps(audio_path, api_key)
# Clean up temporary files
os.unlink(audio_path)
if word_timestamps:
# Format results for display
timestamp_text = ""
for word_info in word_timestamps:
start_time = format_time(word_info["start"])
timestamp_text += f"[{start_time}] {word_info['word']} "
return full_text, timestamp_text
else:
return full_text, None
except Exception as e:
return f"Error processing video: {str(e)}", None
def format_time(seconds):
"""Format time in seconds to MM:SS.ms format."""
minutes = int(seconds // 60)
seconds = seconds % 60
return f"{minutes:02d}:{seconds:05.2f}"
# Create Gradio interface
with gr.Blocks(title="Video Transcription App") as app:
gr.Markdown("# Video Transcription with Word-Level Timestamps")
gr.Markdown("Upload a video file and get word-by-word transcription with timestamps.")
with gr.Row():
with gr.Column():
api_key = gr.Textbox(label="OpenAI API Key", placeholder="Enter your OpenAI API key", type="password")
video_input = gr.Video(label="Upload Video")
submit_btn = gr.Button("Transcribe Video")
with gr.Column():
full_text = gr.Textbox(label="Full Transcription", lines=10)
timestamp_text = gr.Textbox(label="Word-by-Word with Timestamps", lines=15)
submit_btn.click(
fn=process_video,
inputs=[video_input, api_key],
outputs=[full_text, timestamp_text]
)
gr.Markdown("""
## How It Works
1. Upload your video file
2. Enter your OpenAI API key (required for using Whisper API)
3. Click 'Transcribe Video'
4. View the full transcription and word-by-word timestamps
Note: This app extracts audio from your video and sends it to OpenAI's Whisper API for transcription.
Your API key is used only for this request and is not stored.
""")
# Launch the app
if __name__ == "__main__":
app.launch()