video-chat / utils /video_processor.py
corl0s's picture
Intial deploy
e417aa9
Raw
History Blame Contribute Delete
4.4 kB
"""Video processing utilities: frame extraction, audio separation, metadata."""
import base64
import logging
import os
from io import BytesIO
from typing import Optional
import cv2
import ffmpeg
import numpy as np
from PIL import Image
logger = logging.getLogger(__name__)
MAX_FRAME_SIZE = 512
def get_video_metadata(video_path: str) -> dict:
"""Return duration, resolution, fps, frame count, and audio presence for a video file."""
cap = cv2.VideoCapture(video_path)
if not cap.isOpened():
raise ValueError(f"Cannot open video file: {video_path}")
fps = cap.get(cv2.CAP_PROP_FPS) or 0.0
total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
duration_seconds = total_frames / fps if fps > 0 else 0.0
cap.release()
has_audio = False
try:
probe = ffmpeg.probe(video_path)
has_audio = any(s["codec_type"] == "audio" for s in probe.get("streams", []))
except Exception:
logger.warning("Could not probe audio tracks for %s", video_path)
return {
"duration_seconds": duration_seconds,
"fps": fps,
"width": width,
"height": height,
"total_frames": total_frames,
"has_audio": has_audio,
}
def _resize_frame(frame_bgr: np.ndarray, max_size: int = MAX_FRAME_SIZE) -> Image.Image:
"""Resize a BGR numpy frame to fit within max_size×max_size, preserving aspect ratio."""
h, w = frame_bgr.shape[:2]
scale = min(max_size / w, max_size / h, 1.0)
new_w, new_h = int(w * scale), int(h * scale)
resized = cv2.resize(frame_bgr, (new_w, new_h), interpolation=cv2.INTER_AREA)
return Image.fromarray(cv2.cvtColor(resized, cv2.COLOR_BGR2RGB))
def _encode_pil_to_base64(image: Image.Image) -> str:
"""Encode a PIL image as a base64 JPEG string."""
buf = BytesIO()
image.save(buf, format="JPEG", quality=85)
return base64.b64encode(buf.getvalue()).decode("utf-8")
def extract_frames(video_path: str, interval_seconds: int = 5) -> list[dict]:
"""Extract one frame every interval_seconds, resized and base64-encoded.
Returns a list of dicts with keys: frame_index, timestamp_seconds, base64_image.
"""
interval_seconds = int(os.environ.get("FRAME_INTERVAL", interval_seconds))
cap = cv2.VideoCapture(video_path)
if not cap.isOpened():
raise ValueError(f"Cannot open video file: {video_path}")
fps = cap.get(cv2.CAP_PROP_FPS) or 1.0
total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
duration = total_frames / fps
frames = []
frame_index = 0
timestamp = 0.0
while timestamp <= duration:
target_frame = int(timestamp * fps)
cap.set(cv2.CAP_PROP_POS_FRAMES, target_frame)
ret, frame_bgr = cap.read()
if not ret:
break
image = _resize_frame(frame_bgr)
b64 = _encode_pil_to_base64(image)
frames.append({
"frame_index": frame_index,
"timestamp_seconds": timestamp,
"base64_image": b64,
})
logger.info("Extracted frame %d at %.1fs", frame_index, timestamp)
frame_index += 1
timestamp += interval_seconds
cap.release()
logger.info("Extracted %d frames from %s", len(frames), video_path)
return frames
def extract_audio(video_path: str) -> Optional[str]:
"""Extract audio track as a 16kHz mono WAV file saved to /tmp/audio_extracted.wav.
Returns the output path, or None if the video has no audio track.
"""
try:
probe = ffmpeg.probe(video_path)
has_audio = any(s["codec_type"] == "audio" for s in probe.get("streams", []))
except Exception as exc:
logger.warning("ffmpeg probe failed: %s", exc)
return None
if not has_audio:
logger.info("No audio track found in %s", video_path)
return None
output_path = "/tmp/audio_extracted.wav"
try:
(
ffmpeg
.input(video_path)
.output(output_path, acodec="pcm_s16le", ar=16000, ac=1)
.overwrite_output()
.run(quiet=True)
)
logger.info("Audio extracted to %s", output_path)
return output_path
except ffmpeg.Error as exc:
logger.warning("Audio extraction failed: %s", exc.stderr)
return None