"""Video processing utilities: frame extraction, audio separation, metadata.""" import base64 import logging import os from io import BytesIO from typing import Optional import cv2 import ffmpeg import numpy as np from PIL import Image logger = logging.getLogger(__name__) MAX_FRAME_SIZE = 512 def get_video_metadata(video_path: str) -> dict: """Return duration, resolution, fps, frame count, and audio presence for a video file.""" cap = cv2.VideoCapture(video_path) if not cap.isOpened(): raise ValueError(f"Cannot open video file: {video_path}") fps = cap.get(cv2.CAP_PROP_FPS) or 0.0 total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) duration_seconds = total_frames / fps if fps > 0 else 0.0 cap.release() has_audio = False try: probe = ffmpeg.probe(video_path) has_audio = any(s["codec_type"] == "audio" for s in probe.get("streams", [])) except Exception: logger.warning("Could not probe audio tracks for %s", video_path) return { "duration_seconds": duration_seconds, "fps": fps, "width": width, "height": height, "total_frames": total_frames, "has_audio": has_audio, } def _resize_frame(frame_bgr: np.ndarray, max_size: int = MAX_FRAME_SIZE) -> Image.Image: """Resize a BGR numpy frame to fit within max_size×max_size, preserving aspect ratio.""" h, w = frame_bgr.shape[:2] scale = min(max_size / w, max_size / h, 1.0) new_w, new_h = int(w * scale), int(h * scale) resized = cv2.resize(frame_bgr, (new_w, new_h), interpolation=cv2.INTER_AREA) return Image.fromarray(cv2.cvtColor(resized, cv2.COLOR_BGR2RGB)) def _encode_pil_to_base64(image: Image.Image) -> str: """Encode a PIL image as a base64 JPEG string.""" buf = BytesIO() image.save(buf, format="JPEG", quality=85) return base64.b64encode(buf.getvalue()).decode("utf-8") def extract_frames(video_path: str, interval_seconds: int = 5) -> list[dict]: """Extract one frame every interval_seconds, resized and base64-encoded. Returns a list of dicts with keys: frame_index, timestamp_seconds, base64_image. """ interval_seconds = int(os.environ.get("FRAME_INTERVAL", interval_seconds)) cap = cv2.VideoCapture(video_path) if not cap.isOpened(): raise ValueError(f"Cannot open video file: {video_path}") fps = cap.get(cv2.CAP_PROP_FPS) or 1.0 total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) duration = total_frames / fps frames = [] frame_index = 0 timestamp = 0.0 while timestamp <= duration: target_frame = int(timestamp * fps) cap.set(cv2.CAP_PROP_POS_FRAMES, target_frame) ret, frame_bgr = cap.read() if not ret: break image = _resize_frame(frame_bgr) b64 = _encode_pil_to_base64(image) frames.append({ "frame_index": frame_index, "timestamp_seconds": timestamp, "base64_image": b64, }) logger.info("Extracted frame %d at %.1fs", frame_index, timestamp) frame_index += 1 timestamp += interval_seconds cap.release() logger.info("Extracted %d frames from %s", len(frames), video_path) return frames def extract_audio(video_path: str) -> Optional[str]: """Extract audio track as a 16kHz mono WAV file saved to /tmp/audio_extracted.wav. Returns the output path, or None if the video has no audio track. """ try: probe = ffmpeg.probe(video_path) has_audio = any(s["codec_type"] == "audio" for s in probe.get("streams", [])) except Exception as exc: logger.warning("ffmpeg probe failed: %s", exc) return None if not has_audio: logger.info("No audio track found in %s", video_path) return None output_path = "/tmp/audio_extracted.wav" try: ( ffmpeg .input(video_path) .output(output_path, acodec="pcm_s16le", ar=16000, ac=1) .overwrite_output() .run(quiet=True) ) logger.info("Audio extracted to %s", output_path) return output_path except ffmpeg.Error as exc: logger.warning("Audio extraction failed: %s", exc.stderr) return None