from fastapi import FastAPI, HTTPException from fastapi.middleware.cors import CORSMiddleware import requests import re import json from pydantic import BaseModel app = FastAPI() app.add_middleware( CORSMiddleware, allow_origins=["*"], allow_credentials=True, allow_methods=["*"], allow_headers=["*"], ) class TranscriptRequest(BaseModel): url: str class TranslateRequest(BaseModel): text: str targetLang: str def extract_video_id(url: str) -> str: pattern = r'(?:youtube\.com\/(?:[^\/]+\/.+\/|(?:v|e(?:mbed)?)\/|.*[?&]v=)|youtu\.be\/)([^"&?\/\s]{11})' match = re.search(pattern, url) return match.group(1) if match else None def parse_caption_xml(xml_string: str) -> str: text_matches = re.findall(r']*>([^<]*)', xml_string) if not text_matches: return "" full_text = " ".join(text_matches) full_text = full_text.replace("&#39;", "'").replace("&quot;", '"').replace("&amp;", "&").replace("'", "'") return full_text def fetch_via_proxy(proxy_base: str, target_url: str, timeout: int = 15) -> requests.Response: """Fetches a URL through a public CORS/Web proxy.""" encoded_url = requests.utils.quote(target_url, safe='') proxy_url = f"{proxy_base}{encoded_url}" headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36", "Accept": "text/html,application/json" } response = requests.get(proxy_url, headers=headers, timeout=timeout) if response.status_code != 200: raise Exception(f"Proxy returned status {response.status_code}") return response @app.post("/api/transcript") async def get_transcript(req: TranscriptRequest): video_id = extract_video_id(req.url) if not video_id: raise HTTPException(status_code=400, detail="Invalid YouTube URL") # List of reliable public web proxies proxies = [ "https://api.allorigins.win/raw?url=", "https://corsproxy.io/?", "https://api.codetabs.com/v1/proxy?quest=" ] for proxy in proxies: try: print(f"Trying proxy {proxy} for video {video_id}...") # 1. Fetch YouTube Watch Page HTML via Proxy yt_url = f"https://www.youtube.com/watch?v={video_id}" response = fetch_via_proxy(proxy, yt_url, timeout=15) html = response.text # 2. Find the ytInitialPlayerResponse JSON in the HTML match = re.search(r'ytInitialPlayerResponse\s*=\s*(\{.*?\})\s*;', html, re.DOTALL) if not match: print(f"Failed to find player data via {proxy}") continue data = json.loads(match.group(1)) # Check playability playability = data.get("playabilityStatus", {}) if playability.get("status") == "ERROR": raise Exception(playability.get("reason", "Video unavailable")) # 3. Extract caption tracks caption_tracks = data.get("captions", {}).get("playerCaptionsTracklistRenderer", {}).get("captionTracks", []) if not caption_tracks: print(f"No captions found via {proxy}") continue # Prefer English, fallback to first available selected_track = next((t for t in caption_tracks if t.get("languageCode") == "en"), caption_tracks[0]) caption_url = selected_track.get("baseUrl") if not caption_url: continue print(f"Found captions, fetching XML via proxy...") # 4. Fetch the actual caption XML via the same proxy caption_response = fetch_via_proxy(proxy, caption_url, timeout=10) # 5. Parse the XML into plain text full_text = parse_caption_xml(caption_response.text) if not full_text.strip(): continue print(f"✅ Successfully fetched transcript via {proxy}. Length: {len(full_text)} chars") return {"text": full_text} except Exception as e: print(f"❌ Proxy {proxy} failed: {e}") continue raise HTTPException(status_code=500, detail="Failed to fetch transcript. All proxy methods failed or video has no captions.") # ======================================================== # TRANSLATE API # ======================================================== @app.post("/api/translate") async def translate_text(req: TranslateRequest): if not req.text or not req.targetLang: raise HTTPException(status_code=400, detail="Missing text or target language") try: url = f"https://translate.googleapis.com/translate_a/single?client=gtx&sl=auto&tl={req.targetLang}&dt=t" res = requests.post(url, headers={ "Content-Type": "application/x-www-form-urlencoded", "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" }, data=f"q={requests.utils.quote(req.text)}", timeout=15 ) if res.status_code != 200: raise Exception(f"Google API returned {res.status_code}") data = res.json() translated = "".join([item[0] for item in data[0]]) return {"translatedText": translated} except Exception as e: raise HTTPException(status_code=500, detail=f"Translation failed: {str(e)}") # KEEP-ALIVE ENDPOINT @app.get("/api/health") async def health_check(): return {"status": "ok", "message": "Transcript API is running!"}