Cutclip / index.html
Mavya283525's picture
Update index.html
27d6fc4 verified
Raw
History Blame Contribute Delete
11.2 kB
# yt_autoclip_starter/app.py
"""
YouTube Auto Clip (Starter Project)
- Paste a YouTube URL
- Download video
- Auto-detect highlight-worthy clips (simple heuristics)
- Transcribe to captions (Whisper or Faster-Whisper)
- Burn captions with ffmpeg (optional)
- Export MP4 clips and SRT files
- Web UI built with Gradio (works on CPU)
This is an educational starter. You can extend scoring, add AI models, etc.
"""
import os
import re
import subprocess
import shutil
from pathlib import Path
from typing import List, Tuple, Dict
import zipfile
import gradio as gr
# Optional: faster-whisper is faster on CPU; fallback to openai-whisper
USE_FASTER_WHISPER = True
try:
from faster_whisper import WhisperModel # type: ignore
except Exception:
USE_FASTER_WHISPER = False
try:
import whisper # type: ignore
except Exception:
whisper = None
import numpy as np
import ffmpeg # ffmpeg-python
import librosa
from scenedetect import VideoManager, SceneManager
from scenedetect.detectors import ContentDetector
ROOT = Path(__file__).resolve().parent
WORK_DIR = ROOT / "work"
WORK_DIR.mkdir(parents=True, exist_ok=True)
def sanitize_filename(name: str) -> str:
return re.sub(r'[^a-zA-Z0-9_\-\. ]+', '_', name).strip().replace(' ', '_')
def ytdlp_download(url: str, out_dir: Path) -> Tuple[Path, Path, str]:
"""Download the best mp4 and extract audio. Returns (video_path, audio_path, title)."""
out_dir.mkdir(parents=True, exist_ok=True)
temp_template = str(out_dir / "%(title)s.%(ext)s")
cmd = [
"yt-dlp",
"-f", "bestvideo[ext=mp4]+bestaudio[ext=m4a]/best[ext=mp4]/best",
"-o", temp_template,
"--no-playlist",
"--restrict-filenames",
url,
]
subprocess.check_call(cmd)
mp4s = list(out_dir.glob("*.mp4"))
if not mp4s:
raise RuntimeError("Gagal download video MP4. Pastikan link benar & tidak dibatasi.")
video_path = max(mp4s, key=lambda p: p.stat().st_size)
title = video_path.stem
audio_path = out_dir / f"{video_path.stem}.wav"
(
ffmpeg
.input(str(video_path))
.output(str(audio_path), ac=1, ar=16000)
.overwrite_output()
.run(quiet=True)
)
return video_path, audio_path, title
def detect_scenes(video_path: Path, threshold: float = 27.0) -> List[Tuple[int, int]]:
"""Return list of scenes as (start_sec, end_sec)."""
video_manager = VideoManager([str(video_path)])
scene_manager = SceneManager()
scene_manager.add_detector(ContentDetector(threshold=threshold))
video_manager.start()
scene_manager.detect_scenes(frame_source=video_manager)
scene_list = scene_manager.get_scene_list()
fps = video_manager.get_base_timecode().framerate
video_manager.release()
return [(int(start.get_frames()/fps), int(end.get_frames()/fps)) for start, end in scene_list]
def compute_loudness_scores(audio_path: Path, frame_sec: float = 1.0) -> np.ndarray:
"""Compute per-second RMS loudness as a simple excitement proxy."""
y, sr = librosa.load(str(audio_path), sr=None, mono=True)
hop = int(frame_sec * sr)
rms = librosa.feature.rms(y=y, frame_length=hop, hop_length=hop).flatten()
if rms.size == 0:
return np.zeros(1)
rms = (rms - rms.min()) / (rms.max() - rms.min() + 1e-9)
return rms
def propose_clips(scenes: List[Tuple[int, int]], loud: np.ndarray,
max_len: int = 30, min_len: int = 7) -> List[Tuple[int, int, float]]:
"""Propose candidate clips based on scene boundaries & loudness peaks."""
candidates = []
for s, e in scenes:
length = e - s
if length < min_len:
continue
if length > max_len:
for start in range(s, e, max_len):
end = min(start + max_len, e)
score = float(loud[start:end].mean()) if end < len(loud) else 0.0
candidates.append((start, end, score))
else:
score = float(loud[s:e].mean()) if e < len(loud) else 0.0
candidates.append((s, e, score))
candidates.sort(key=lambda x: x[2], reverse=True)
return candidates
def transcribe(audio_path: Path, model_size: str = "small") -> List[Dict]:
"""Transcribe audio and return list of segments {start, end, text}."""
segments = []
if USE_FASTER_WHISPER:
model = WhisperModel(model_size, device="cpu", compute_type="int8")
segs, _ = model.transcribe(str(audio_path), vad_filter=True)
for s in segs:
segments.append({"start": s.start, "end": s.end, "text": s.text.strip()})
else:
if whisper is None:
return segments
model = whisper.load_model(model_size)
result = model.transcribe(str(audio_path))
for s in result.get("segments", []):
segments.append({"start": s["start"], "end": s["end"], "text": s["text"].strip()})
return segments
def segments_to_srt(segments: List[Dict]) -> str:
def fmt_time(t: float) -> str:
h = int(t // 3600); t -= h*3600
m = int(t // 60); t -= m*60
s = int(t); ms = int((t - s) * 1000)
return f"{h:02}:{m:02}:{s:02},{ms:03}"
lines = []
for i, seg in enumerate(segments, 1):
lines.append(str(i))
lines.append(f"{fmt_time(seg['start'])} --> {fmt_time(seg['end'])}")
lines.append(seg['text'])
lines.append("")
return "\n".join(lines)
def clip_video(video_path: Path, out_path: Path, start: int, end: int,
srt_path: Path | None = None, burn_captions: bool = False):
inp = ffmpeg.input(str(video_path), ss=start, to=end)
if burn_captions and srt_path is not None and srt_path.exists():
out = ffmpeg.output(
inp.video.filter_("subtitles", str(srt_path)),
inp.audio, str(out_path),
vcodec="libx264", acodec="aac", movflags="+faststart"
)
else:
out = ffmpeg.output(
inp.video, inp.audio, str(out_path),
vcodec="libx264", acodec="aac", movflags="+faststart"
)
out.overwrite_output().run(quiet=True)
def align_captions_to_clip(segments: List[Dict], start: int, end: int) -> List[Dict]:
out = []
for seg in segments:
if seg["end"] < start or seg["start"] > end:
continue
s = max(seg["start"], start) - start
e = min(seg["end"], end) - start
out.append({"start": max(0.0, s), "end": max(0.0, e), "text": seg["text"]})
return out
def viral_score_heuristic(text: str, score: float) -> float:
hooks = ["cara", "tips", "rahasia", "penting", "jangan", "kenapa", "fakta", "wow", "tutorial"]
bonus = 0.0
low = text.lower()
for h in hooks:
if h in low:
bonus += 0.05
return min(1.0, score * 0.8 + bonus)
def process_url(url: str, max_candidates: int = 8, model_size: str = "small", burn_subs: bool = True):
session_dir = WORK_DIR / sanitize_filename(url)
if session_dir.exists():
shutil.rmtree(session_dir)
session_dir.mkdir(parents=True, exist_ok=True)
video_path, audio_path, title = ytdlp_download(url, session_dir)
scenes = detect_scenes(video_path)
loud = compute_loudness_scores(audio_path)
candidates = propose_clips(scenes, loud)
segs = transcribe(audio_path, model_size=model_size)
results = []
export_dir = session_dir / "exports"
export_dir.mkdir(exist_ok=True)
for i, (s, e, base_score) in enumerate(candidates[:max_candidates], start=1):
clip_txt_segs = align_captions_to_clip(segs, s, e)
srt_txt = segments_to_srt(clip_txt_segs) if clip_txt_segs else ""
clip_name = f"{sanitize_filename(title)}_clip_{i:02d}_{s:04d}-{e:04d}.mp4"
clip_path = export_dir / clip_name
srt_path = export_dir / (clip_name.replace(".mp4", ".srt"))
if srt_txt:
srt_path.write_text(srt_txt, encoding="utf-8")
try:
clip_video(video_path, clip_path, s, e, srt_path if srt_txt else None, burn_captions=burn_subs)
except ffmpeg.Error:
clip_video(video_path, clip_path, s, e, None, burn_captions=False)
joined_text = " ".join([x["text"] for x in clip_txt_segs]) if clip_txt_segs else ""
vscore = viral_score_heuristic(joined_text, base_score)
results.append({
"clip": str(clip_path.relative_to(ROOT)),
"srt": str(srt_path.relative_to(ROOT)) if srt_txt else "",
"start": s, "end": e,
"score": round(float(vscore), 3),
"words": joined_text[:200] + ("..." if joined_text and len(joined_text) > 200 else ""),
})
return results, str(export_dir.relative_to(ROOT))
def ui_process(url, max_candidates, model_size, burn_subs):
try:
results, export_dir = process_url(url, int(max_candidates), model_size, bool(burn_subs))
table = [
[r["clip"], r["srt"], f"{r['start']}–{r['end']} s", r["score"], r["words"]]
for r in results
]
return export_dir, table, gr.update(choices=[r["clip"] for r in results], value=[r["clip"] for r in results[:3]])
except Exception as e:
return "", [["ERROR", "", "", "", str(e)]], gr.update(choices=[], value=[])
def zip_selected_clips(selected: List[str]):
if not selected:
return None
zip_path = ROOT / "downloads" / "selected_clips.zip"
zip_path.parent.mkdir(exist_ok=True)
with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as zf:
for rel in selected:
abs_path = ROOT / rel
if abs_path.exists():
zf.write(abs_path, abs_path.name)
srt = abs_path.with_suffix(".srt")
if srt.exists():
zf.write(srt, srt.name)
return str(zip_path)
with gr.Blocks(title="YouTube Auto Clip • Starter", css="footer{visibility:hidden}") as demo:
gr.Markdown("# 🎬 YouTube Auto Clip (Starter)\nPaste a YouTube URL → get auto-suggested clips + captions.\n\n> ⚠️ For demo/education. Respect YouTube Terms & creator rights.")
with gr.Row():
url = gr.Textbox(label="YouTube URL", placeholder="https://www.youtube.com/watch?v=...")
with gr.Row():
max_candidates = gr.Slider(2, 20, value=8, step=1, label="Jumlah Kandidat Klip")
model_size = gr.Dropdown(["tiny", "base", "small", "medium", "large-v3"], value="small", label="Model Transkripsi (Whisper)")
burn_subs = gr.Checkbox(value=True, label="Bakar Subtitle (Burn-in)")
run_btn = gr.Button("▶️ Proses")
export_dir = gr.Textbox(label="Folder Ekspor", interactive=False)
results_tbl = gr.Dataframe(headers=["MP4 Path", "SRT Path", "Durasi (s)", "Skor Viral", "Cuplikan Teks"], wrap=True)
selected = gr.CheckboxGroup(choices=[], label="Pilih klip untuk diunduh (ikut SRT)")
zip_btn = gr.Button("📦 ZIP klip terpilih")
zip_file = gr.File(label="Unduh ZIP", interactive=False)
run_btn.click(ui_process, inputs=[url, max_candidates, model_size, burn_subs], outputs=[export_dir, results_tbl, selected])
zip_btn.click(zip_selected_clips, inputs=[selected], outputs=[zip_file])
if __name__ == "__main__":
demo.launch()