diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000000000000000000000000000000000000..40f60e1de89e0f904c11d302058de8c039f587a9 --- /dev/null +++ b/.gitignore @@ -0,0 +1,6 @@ +__pycache__/ +*.pyc +data/ +.data +.huggingface/ +.restart_trigger diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000000000000000000000000000000000000..bed2a391f57aae4c106228b98824a5a07c71913b --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,16 @@ +# FPT Play Stream Selector Update + +Added FPT Play channel with stream selector UI similar to VTV6: + +- New tab "FPT" (orange themed) in the channel tabs +- Stream selector with 4 sources: + 1. 🌐 Web FPT Play (iframe) + 2. 📡 HLS Proxy (via /api/proxy/m3u8) + 3. 🔗 HLS Direct + 4. 📺 HD1.xemtv.net (iframe from LINK 1) +- Backend vtv_api.py now returns stream_selectors for fpt-the-thao channel +- Frontend handles stream switching automatically when FPT tab is active + +## Changes +- `static/vtv_init.js`: Added FPT tab + stream selector UI logic +- `vtv_api.py`: Added FPT Play endpoint responses with stream_selectors data diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000000000000000000000000000000000000..fd799e0b9eca048c628530076b0a14544944a46b --- /dev/null +++ b/Dockerfile @@ -0,0 +1,48 @@ +FROM python:3.12-slim + +WORKDIR /app + +RUN echo "[BUILD] step1: apt-get update+install ffmpeg + Vietnamese fonts" && \ + apt-get update && apt-get install -y --no-install-recommends \ + ffmpeg \ + fonts-dejavu-core \ + fonts-noto \ + fonts-noto-cjk \ + fonts-noto-color-emoji \ + fonts-liberation \ + fonts-freefont-ttf \ + libfreetype6 \ + && rm -rf /var/lib/apt/lists/* && \ + echo "[BUILD] step1 done" + +RUN echo "[BUILD] step2: pip base pkgs (bs4/lxml)" && \ + pip install --no-cache-dir "beautifulsoup4>=4.12" lxml && \ + echo "[BUILD] step2 done" + +RUN echo "[BUILD] step3: pip main pkgs" && \ + pip install --no-cache-dir fastapi uvicorn requests beautifulsoup4 jinja2 yt-dlp huggingface_hub gTTS pillow edge-tts python-dateutil httpx pycryptodome && \ + echo "[BUILD] step3 done" + +COPY requirements.txt . +RUN echo "[BUILD] step4: pip requirements.txt" && \ + pip install --no-cache-dir -r requirements.txt || true && \ + echo "[BUILD] step4 done" + +COPY . . +EXPOSE 7860 + +RUN echo "[BUILD] step5: setup Vietnamese font symlink" && \ + mkdir -p /usr/share/fonts/truetype/vn && \ + # Prefer Noto Sans for Vietnamese - it has full diacritic support + if [ -f /usr/share/fonts/truetype/noto/NotoSans-Regular.ttf ]; then \ + ln -sf /usr/share/fonts/truetype/noto/NotoSans-Regular.ttf /usr/share/fonts/truetype/vn/VNFont.ttf; \ + elif [ -f /usr/share/fonts/truetype/dejavu/DejaVuSans.ttf ]; then \ + ln -sf /usr/share/fonts/truetype/dejavu/DejaVuSans.ttf /usr/share/fonts/truetype/vn/VNFont.ttf; \ + ln -sf /usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf /usr/share/fonts/truetype/vn/VNFont-Bold.ttf; \ + fi; \ + fc-cache -f -v || true; \ + date > /app/.build_done && \ + echo "[BUILD] step5 done" + +CMD ["uvicorn", "_run:app", "--host", "0.0.0.0", "--port", "7860"] +# v3.0-vn-font-fix-short-video-2026-07-19 diff --git a/README.md b/README.md new file mode 100644 index 0000000000000000000000000000000000000000..f65ec063b0184f66ec5c41d94307c4f38f4c0a38 --- /dev/null +++ b/README.md @@ -0,0 +1,58 @@ +--- +title: VNEWS +emoji: 📰 +colorFrom: green +colorTo: yellow +sdk: docker +pinned: false +tags: +- ml-intern +--- + +# VNEWS - Tin Tức Việt Nam + +**v18 - FIXED VTV2/VTV3/VTV6/VTV9 stream hanging** + +## 🔧 Changes in v18 (2026-07-06) +- **VTV2, VTV3, VTV6, VTV9**: Skip expired ssaimh CDN token → immediately fall through to sv2.xemtivitop.com +- **15+ extraction patterns** for m3u8 URL (up from 5), including: file:, src=, source:, player.src(), hls.loadSource(), href=, ``, url:, window.location, iframe follow (3 levels deep), base64 decode +- **Backup CDN** `tv.mediacdn.vn` for VTV2/VTV3/VTV6/VTV9 +- **Fast timeout** 5s for CDN, 12s for PHP endpoints (was 15s each = 60s+ total) +- **sv2.xemtivitop.com** re-prioritized to check BEFORE xemtv.us +- **Iframe chain following**: if a PHP page returns an iframe → follow it up to 3 levels to find the m3u8 + +## Features: +- 📰 News from VnExpress (10 categories) + GenK AI +- ⚽ Livescore from bongda.com.vn (live, today, upcoming, results, standings) +- 🎬 Football highlights from xemlaibongda.top (8 leagues) +- 📺 VTV live channels (VTV1→VTV10, VTV Prime) + - Priority: ssaimh CDN → sv2.xemtivitop.com → xemtv.us → xemtivitop blogspot → FPTPlay → VTVGo → mediacdn → xemtv.net +- 🏆 World Cup 2026 (news, fixtures, standings, stats, highlights) +- 🤖 AI article writing + TTS (multilingual, emotion-aware) +- 🔍 Topic search (8 news sources) +- 🎤 TTS: voice selector + emotion selector + speed control + +## 🎬 Short AI — Video từ link (scrap YouTube / TikTok / tin tức) + +Short creator có chế độ **"🔗 Video từ link"**: dán link video YouTube / TikTok / +VnExpress / Dân trí / Znews / 24h... → bấm "Lấy video" để xem trước → tạo short +chạy video + ảnh đã chọn bù phần còn thiếu nếu video ngắn hơn giọng đọc. + +### 🔑 Cài cookies cho YouTube (bỏ chặn "Sign in to confirm you're not a bot") +YouTube đôi khi chặn IP datacenter. Cách khắc phục bằng cookies: + +1. Cài extension trình duyệt **"Get cookies.txt LOCALLY"** (Chrome/Edge) hoặc + **"cookies.txt"** (Firefox). +2. Mở `https://www.youtube.com` (đã đăng nhập) → bấm extension → **Export** → ra file `cookies.txt` (định dạng Netscape). +3. Đưa cookies vào Space bằng **một trong hai cách**: + - **Cách A (khuyến nghị):** Vào Settings của Space + `huggingface.co/spaces/bep40/VNEWS/settings` → **Variables and secrets** → + tạo secret tên **`YT_COOKIES`**, giá trị = toàn bộ nội dung file `cookies.txt`. + - **Cách B:** đặt file `cookies.txt` vào thư mục gốc repo `VNEWS/` và commit + (chú ý: cookies sẽ công khai nếu repo public — ưu tiên Cách A). +4. Rebuild Space (mỗi lần đổi secret phải **Restart** Space). + +Backend tự đọc `YT_COOKIES` (secret) hoặc `/app/cookies.txt`, ghi thành file tạm +và truyền cho yt-dlp qua `cookiefile`. Không cần sửa code. + +> Lưu ý: cookies có hạn (thường vài tuần). Khi hết hạn, export lại và cập nhật secret. \ No newline at end of file diff --git a/_run.py b/_run.py new file mode 100644 index 0000000000000000000000000000000000000000..de72380c3006dba490ec49c1fe421c87723a0490 --- /dev/null +++ b/_run.py @@ -0,0 +1 @@ +from app_v2_entry import app # v5-stable inline bongda proxy \ No newline at end of file diff --git a/ai_ext.py b/ai_ext.py new file mode 100644 index 0000000000000000000000000000000000000000..ab7540db0765711cf34947d481888084e0f87c65 --- /dev/null +++ b/ai_ext.py @@ -0,0 +1,332 @@ +"""VNEWS AI Extension - rewrite + auto short video generation. +Imported by app_v2_entry.py to register /api/rewrite_share, /api/topic_post, +/api/ai_wall, /api/wall, /api/ai/short endpoints on the main FastAPI app. + +Uses main.py's WALL_FILE (wall_posts.json) for unified data store. +TTS: edge-tts (HoaiMy female, NamMinh male) with speed control + gTTS fallback. +""" +import os, re, json, time, random, html as html_lib, subprocess, asyncio +from urllib.parse import quote_plus, quote, urlparse, urljoin +from typing import Optional, List, Dict +import requests +from bs4 import BeautifulSoup +from fastapi import Request, Query +from fastapi.responses import HTMLResponse, JSONResponse, FileResponse + +# Try to import main app, but don't fail if it doesn't exist +try: + from main import app +except ImportError: + # Create a minimal FastAPI app for standalone testing + try: + from fastapi import FastAPI + app = FastAPI() + except Exception: + app = None + +# Import wall store from main.py so we read/write the SAME file +try: + from main import _load_wall, _save_wall, _web_context # noqa: F401 +except ImportError: + _data_dir = "/data" if os.path.isdir("/data") else "/app/data" + _wall_file = os.path.join(_data_dir, "wall_posts.json") + def _load_wall(): + try: + if os.path.exists(_wall_file): + with open(_wall_file, "r", encoding="utf-8") as f: + return json.load(f) + except Exception: + pass + return [] + def _save_wall(posts): + try: + os.makedirs(os.path.dirname(_wall_file), exist_ok=True) + tmp = _wall_file + ".tmp" + with open(tmp, "w", encoding="utf-8") as f: + json.dump(posts[:100], f, ensure_ascii=False) + os.replace(tmp, _wall_file) + except Exception: + pass + def _web_context(topic): + return "" + +# ai_ext alias for backward compatibility +_load_ai_wall = _load_wall +_save_ai_wall = _save_wall + +try: + from huggingface_hub import AsyncInferenceClient +except Exception: + AsyncInferenceClient = None +try: + from gtts import gTTS +except Exception: + gTTS = None +try: + from PIL import Image, ImageDraw, ImageFont +except Exception: + Image = ImageDraw = ImageFont = None +try: + import edge_tts +except Exception: + edge_tts = None + + +def _hf_token(): + for k in ("HF_TOKEN", "HUGGINGFACE_HUB_API_TOKEN", "HUGGING_FACE_HUB_TOKEN", "HF_API_TOKEN"): + v = os.getenv(k, "").strip() + if v: + return v + return "" + + +def _clean_text(s: str) -> str: + """Clean text for processing.""" + s = html_lib.unescape(s or "") + s = re.sub(r"\s+", " ", s) + return s.strip() + + +def _domain(url: str) -> str: + """Extract domain from URL.""" + try: + return urlparse(url or "").netloc.replace("www.", "") + except Exception: + return "" + + +async def qwen_generate(prompt: str, image_url: str = None, max_tokens: int = 1200) -> str: + """Generate text using Llama/Qwen models via Hugging Face Inference API. + + Prioritizes Llama-3.3-70B for better creative/opinion writing. + """ + token = _hf_token() + errors = [] + + # Try HF router API with multiple models - Llama FIRST for opinion writing + if token: + models = [ + os.getenv("QWEN_VL_MODEL", ""), + "meta-llama/Llama-3.3-70B-Instruct", # FIRST - best for opinion/analysis + "Qwen/Qwen2.5-VL-7B-Instruct", + "Qwen/Qwen2.5-72B-Instruct", + ] + # Deduplicate while preserving order + seen = set() + models = [m for m in models if m and m not in seen and not seen.add(m)] + + headers = {"Authorization": f"Bearer {token}", "Content-Type": "application/json"} + + for model in models: + try: + is_vl = "VL" in model and image_url + if is_vl: + user_content = [ + {"type": "image_url", "image_url": {"url": image_url}}, + {"type": "text", "text": prompt} + ] + else: + user_content = prompt + + payload = { + "model": model, + "messages": [ + {"role": "system", "content": "Bạn là nhà báo phản biện chuyên nghiệp. Luôn viết theo quan điểm cá nhân, phân tích sâu, không sao chép nguyên văn nguồn tin."}, + {"role": "user", "content": user_content}, + ], + "max_tokens": min(int(max_tokens or 2000), 2500), + "temperature": 0.75, + "top_p": 0.9, + } + + r = requests.post( + "https://router.huggingface.co/v1/chat/completions", + headers=headers, + json=payload, + timeout=95 + ) + + if r.status_code >= 300: + errors.append(f"{model}: HTTP {r.status_code}") + continue + + j = r.json() + txt = (j.get("choices", [{}])[0].get("message", {}).get("content") or "").strip() + + if txt: + return txt + + errors.append(f"{model}: empty response") + + except Exception as e: + errors.append(f"{model}: {type(e).__name__}") + + # Fallback: extractive summary from prompt + LAST_QWEN_ERROR = errors[-3:] if errors else "unknown error" + return _fallback_summary_from_prompt(prompt, max_units=6) + + +def _fallback_summary_from_prompt(prompt: str, max_units: int = 6) -> str: + """Generate a simple fallback summary when AI is unavailable.""" + text = prompt or "" + for marker in ["Nội dung nguồn:", "Nội dung bài:", "Nội dung gốc:", "Nội dung:", "Nguồn/bối cảnh internet:"]: + if marker in text: + text = text.split(marker, 1)[1] + break + text = re.sub(r"https?://\S+", "", text) + text = re.sub(r"\s+", " ", text).strip() + + # Split into sentences + sentences = re.split(r"(?<=[.!?])\s+(?=[A-ZÀ-Ỹ0-9])", text) + units = [] + for s in sentences: + s = _clean_text(s) + if len(s) >= 30: + units.append(s) + + if units: + result_units = units[:max_units] + return "\n".join("• " + u for u in result_units) + if text: + chunks = [] + for i in range(0, min(len(text), max_units * 300), 280): + chunk = _clean_text(text[i:i+300]) + if chunk and chunk not in chunks: + chunks.append(chunk) + if len(chunks) >= max_units: + break + if chunks: + return "\n".join("• " + c for c in chunks) + return "• Không có đủ nội dung để tóm tắt." + +# ===== URL scraping & article processing ===== +HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36", "Accept-Language": "vi-VN,vi;q=0.9,en;q=0.8"} + +try: + _shorts_base = "/data" if os.path.isdir("/data") else os.path.join(os.path.dirname(os.path.abspath(__file__)), "data") +except Exception: + _shorts_base = os.path.join(os.path.dirname(os.path.abspath(__file__)), "data") +SHORTS_DIR = os.path.join(_shorts_base, "ai_shorts") +os.makedirs(SHORTS_DIR, exist_ok=True) + +import random as _random2 +from datetime import datetime, timezone, timedelta +_VN_TZ = timezone(timedelta(hours=7)) + + +def _safe_name(filename: str) -> str: + """Sanitize filename.""" + return re.sub(r"[^a-zA-Z0-9_.-]", "_", filename)[:120] + + +def pollinations_image_url(topic: str) -> str: + """Generate a placeholder image URL via Pollinations.""" + try: + return "https://image.pollinations.ai/prompt/" + quote("Vietnamese editorial illustration, " + topic, safe="") + "?width=1024&height=576&nologo=true" + except Exception: + return "" + + +def _download_image(url: str, fallback_title: str, out_path: str) -> str: + """Download an image from URL or create a placeholder.""" + if url: + try: + r = requests.get(url, headers=HEADERS, timeout=15) + if r.status_code == 200 and len(r.content) > 1200: + os.makedirs(os.path.dirname(out_path), exist_ok=True) + with open(out_path, "wb") as f: + f.write(r.content) + return out_path + except Exception: + pass + # Fallback: create a placeholder image + try: + from PIL import Image, ImageDraw, ImageFont + img = Image.new("RGB", (1080, 760), (24, 24, 24)) + draw = ImageDraw.Draw(img) + try: + font = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf", 48) + except Exception: + font = None + text = (fallback_title or "VNEWS")[:40] + try: + bbox = draw.textbbox((0, 0), text, font=font) + tw = bbox[2] - bbox[0] + except Exception: + tw = len(text) * 24 + draw.text(((1080 - tw) // 2, 330), text, fill=(255, 255, 255), font=font) + os.makedirs(os.path.dirname(out_path), exist_ok=True) + img.save(out_path, quality=90) + return out_path + except Exception: + return out_path + + +def scrape_any_url(url: str) -> dict: + """Scrape article content from any URL.""" + if not url or not url.startswith("http"): + return {"title": "", "text": "", "summary": "", "image": "", "og_image": "", "via": ""} + try: + r = requests.get(url, headers=HEADERS, timeout=15, allow_redirects=True) + if r.status_code != 200 or not r.text: + return {"title": "", "text": "", "summary": "", "image": "", "og_image": "", "via": _domain(url)} + r.encoding = "utf-8" + soup = BeautifulSoup(r.text, "lxml") + for tag in soup.find_all(["script", "style", "nav", "footer", "aside", "form", "noscript", "iframe", ".ads", ".ad", ".banner-ads", ".fb-comments", ".fb-root", ".social-share", ".related-news", ".breadcrumb"]): + tag.decompose() + title = "" + ogt = soup.find("meta", property="og:title") + if ogt: + title = ogt.get("content", "") + h1 = soup.find("h1") + if not title and h1: + title = h1.get_text(strip=True) + if not title: + t = soup.find("title") + if t: + title = t.get_text(strip=True) + og_image = "" + ogi = soup.find("meta", property="og:image") + if ogi: + og_image = ogi.get("content", "") + if og_image.startswith("//"): + og_image = "https:" + og_image + summary = "" + ogd = soup.find("meta", property="og:description") or soup.find("meta", attrs={"name": "description"}) + if ogd: + summary = ogd.get("content", "")[:500] + body_text = [] + for sel in ["article", ".singular-content", ".detail-content", ".fck_detail", ".content-detail", ".knc-content", "main", ".cms-body", ".article__body", ".post-content", ".entry-content"]: + el = soup.select_one(sel) + if el and len(el.find_all("p")) >= 2: + for p in el.find_all("p"): + t = _clean_text(p.get_text(strip=True)) + if t and len(t) > 30: + body_text.append(t) + break + if not body_text and soup.body: + for p in soup.body.find_all("p"): + t = _clean_text(p.get_text(strip=True)) + if t and len(t) > 30: + body_text.append(t) + text = "\n".join(body_text) + return {"title": _clean_text(title), "text": text, "summary": _clean_text(summary), "image": og_image, "og_image": og_image, "via": _domain(url), "url": url} + except Exception as e: + return {"title": "", "text": "", "summary": "", "image": "", "og_image": "", "via": _domain(url)} + + +def make_post(title: str, text: str, img: str, url: str, kind: str = "auto", sources: list = None) -> dict: + """Create a wall post dict.""" + import random as _r2 + now = int(time.time() * 1000) + return { + "id": str(now) + str(_r2.randint(100, 999)), + "title": (title or "Bài viết")[:200], + "text": (text or "")[:5000], + "img": img or "", + "url": url or "", + "kind": kind or "auto", + "sources": sources or [], + "created": now, + "created_str": datetime.now(_VN_TZ).strftime("%H:%M %d/%m/%Y"), + } diff --git a/ai_fix2.py b/ai_fix2.py new file mode 100644 index 0000000000000000000000000000000000000000..895be1d7505ed5fedaec5b8f023a52434101989f --- /dev/null +++ b/ai_fix2.py @@ -0,0 +1,366 @@ +import os, re, subprocess, html as html_lib, json +from urllib.parse import quote_plus, urlparse, parse_qs, unquote +import requests +import ai_patch as prev +from ai_patch import app +from fastapi import Request +from fastapi.responses import JSONResponse, HTMLResponse, FileResponse + +base = prev.base + + +def clean(s): + return re.sub(r"\s+", " ", html_lib.unescape(s or "")).strip() + + +def _is_real_article_text(raw): + raw = clean(raw) + if len(raw) < 500: + return False + # Reject search-result/title-only pages: need several real sentences. + sentences = re.split(r"(?<=[\.\!\?])\s+", raw) + long_sentences = [s for s in sentences if len(s) > 45] + return len(long_sentences) >= 5 + + +def _extract_ddg_url(href): + if not href: + return "" + if href.startswith("//"): + href = "https:" + href + if "duckduckgo.com/l/" in href: + try: + qs = parse_qs(urlparse(href).query) + if qs.get("uddg"): + return unquote(qs["uddg"][0]) + except Exception: + pass + return href + + +def _ddg_article_urls(topic, limit=12): + urls = [] + try: + q = quote_plus(topic + " tin tức bài viết phân tích") + r = requests.get("https://html.duckduckgo.com/html/?q=" + q, headers=base.HEADERS, timeout=18) + r.encoding = "utf-8" + from bs4 import BeautifulSoup + soup = BeautifulSoup(r.text, "lxml") + for a in soup.select("a.result__a"): + u = _extract_ddg_url(a.get("href", "")) + if not u.startswith("http"): + continue + if any(bad in u for bad in ["google.com", "youtube.com", "facebook.com", "x.com", "twitter.com"]): + continue + if u not in urls: + urls.append(u) + if len(urls) >= limit: + break + except Exception: + pass + return urls + + +def _rss_article_urls(topic, limit=10): + out = [] + try: + url = "https://news.google.com/rss/search?q=" + quote_plus(topic) + "&hl=vi&gl=VN&ceid=VN:vi" + r = requests.get(url, headers=base.HEADERS, timeout=15) + r.encoding = "utf-8" + from bs4 import BeautifulSoup + soup = BeautifulSoup(r.text, "xml") + for it in soup.find_all("item")[:limit]: + title = it.find("title").get_text(" ", strip=True) if it.find("title") else "" + link = it.find("link").get_text(strip=True) if it.find("link") else "" + src = it.find("source").get_text(" ", strip=True) if it.find("source") else base._domain(link) + if title and link: + out.append({"title": title, "url": link, "via": src, "excerpt": title}) + except Exception: + pass + return out + + +def _topic_source_articles(topic, limit=5): + """Scrape actual article bodies. Do not accept title-only sources.""" + candidates = [] + seen = set() + + # 1) DuckDuckGo actual result URLs are usually more directly scrapable. + for u in _ddg_article_urls(topic, limit=14): + if u not in seen: + seen.add(u) + candidates.append({"url": u, "title": "", "via": base._domain(u)}) + + # 2) Add base web_context sources. + try: + _ctx, srcs = base.web_context(topic, limit=8) + for s in srcs or []: + u = s.get("url") or "" + if u.startswith("http") and u not in seen: + seen.add(u) + candidates.append(s) + except Exception: + pass + + # 3) Google News RSS fallback last. + for s in _rss_article_urls(topic, limit=10): + u = s.get("url") or "" + if u.startswith("http") and u not in seen: + seen.add(u) + candidates.append(s) + + out = [] + for s in candidates[:24]: + url = s.get("url") or "" + try: + page = base.scrape_any_url(url) + raw = (page.get("summary", "") + "\n" + page.get("text", "")).strip() + if not _is_real_article_text(raw): + continue + title = page.get("title") or s.get("title") or url + via = page.get("via") or s.get("via") or base._domain(url) + out.append({ + "title": title, + "url": url, + "raw": raw, + "image": page.get("image") or "", + "via": via, + "source": {"title": title, "url": url, "excerpt": raw[:700], "via": via} + }) + if len(out) >= limit: + break + except Exception: + continue + return out[:limit] + + +def sentence_split(text): + text = re.sub(r"^[•\-\*]\s*", "", text or "", flags=re.M) + text = re.sub(r"\n+", ". ", text) + parts = [] + for s in re.split(r"(?<=[\.\!\?])\s+", text): + s = clean(s) + if len(s) >= 8: + parts.append(s) + return parts + + +def srt_time(sec): + ms = int((sec - int(sec)) * 1000) + sec = int(sec) + return f"{sec//3600:02d}:{(sec%3600)//60:02d}:{sec%60:02d},{ms:03d}" + + +def parse_timecode(t): + # 00:00:01.234 or 00:00:01,234 + t = t.replace(',', '.') + parts = t.split(':') + if len(parts) == 3: + return int(parts[0])*3600 + int(parts[1])*60 + float(parts[2]) + if len(parts) == 2: + return int(parts[0])*60 + float(parts[1]) + return float(parts[0]) + + +def convert_vtt_to_scaled_srt(vtt_path, srt_path, speed=1.2): + try: + txt = open(vtt_path, 'r', encoding='utf-8').read().splitlines() + cues = [] + i = 0 + while i < len(txt): + line = txt[i].strip() + if '-->' in line: + a, b = [x.strip().split()[0] for x in line.split('-->')[:2]] + start = parse_timecode(a) / speed + end = parse_timecode(b) / speed + i += 1 + texts = [] + while i < len(txt) and txt[i].strip(): + texts.append(txt[i].strip()) + i += 1 + s = clean(' '.join(texts)) + if s: + cues.append((start, end, s)) + i += 1 + if not cues: + return False + with open(srt_path, 'w', encoding='utf-8') as f: + for idx, (st, en, s) in enumerate(cues, 1): + if en <= st: + en = st + 1.2 + f.write(f"{idx}\n{srt_time(st)} --> {srt_time(en)}\n{s}\n\n") + return True + except Exception: + return False + + +def write_weighted_srt(script, path, total_duration): + subs = sentence_split(script) + if not subs: + subs = [clean(script)[:140] or "VNEWS"] + total_chars = max(1, sum(len(x) for x in subs)) + usable = max(2.0, float(total_duration) - 1.0) + cur = 0.5 + with open(path, "w", encoding="utf-8") as f: + for i, s in enumerate(subs, 1): + dur = max(1.8, min(7.0, usable * len(s) / total_chars)) + start = cur + end = min(total_duration - 0.15, cur + dur) + cur = end + 0.18 + f.write(f"{i}\n{srt_time(start)} --> {srt_time(end)}\n{s}\n\n") + if cur >= total_duration - 0.2: + break + + +def tts_script_full(post, emotion): + title = clean(post.get("title", "")) + text = clean(post.get("text", "")) + text = re.sub(r"Nguồn tham khảo:.*", "", text, flags=re.S).strip() + prefix = { + "urgent": "Tin nhanh.", + "warm": "Câu chuyện đáng chú ý.", + "serious": "Bản tin nghiêm túc.", + "energetic": "Cập nhật nổi bật.", + }.get(emotion, "") + script = f"{prefix} {title}. {text}".strip() + # Keep complete wall summary. Only trim pathological payloads, on sentence boundary. + if len(script) > 3600: + tmp = script[:3600] + cut = max(tmp.rfind("."), tmp.rfind("!"), tmp.rfind("?")) + script = tmp[:cut + 1] if cut > 1600 else tmp + script = re.sub(r"([\.\!\?])\s*", r"\1\n", script) + script = re.sub(r"\n{2,}", "\n", script).strip() + return script + + +_PATCH = {('/api/topic_post','POST'),('/api/ai/short/{post_id}','POST'),('/api/ai/short-file/{file_id}','GET'),('/','GET')} +app.router.routes = [r for r in app.router.routes if not any(getattr(r,'path',None)==p and m in getattr(r,'methods',set()) for p,m in _PATCH)] + + +@app.post('/api/topic_post') +async def topic_post_aggregate(request: Request): + body = await request.json() + topic = base._clean_text(body.get('topic','')) + if not topic: + return JSONResponse({'error':'missing topic'}, status_code=400) + articles = _topic_source_articles(topic, limit=5) + if not articles: + return JSONResponse({'error':'Không scrape được nội dung bài viết thật cho chủ đề này. Hãy thử chủ đề cụ thể hơn hoặc dán URL trực tiếp.'}, status_code=422) + source_blocks = [] + sources = [] + image = "" + for i, art in enumerate(articles, 1): + raw = art.get('raw','') + source_blocks.append(f"[Nguồn {i}] {art.get('title','')} ({art.get('via','')})\n{raw[:3000]}") + sources.append(art.get('source') or {'title': art.get('title'), 'url': art.get('url'), 'via': art.get('via'), 'excerpt': raw[:600]}) + if not image and art.get('image'): + image = art.get('image') + ctx = "\n\n".join(source_blocks) + prompt = f"""Bạn là biên tập viên tổng hợp tin tức tiếng Việt. + +Chủ đề: {topic} + +NHIỆM VỤ: +- Đọc nội dung của TẤT CẢ các bài nguồn bên dưới. +- Tổng hợp thành 1 bản tóm tắt chung duy nhất, giống cách tóm tắt qua URL. +- Không tạo mỗi tiêu đề thành một bài riêng. +- Không chỉ liệt kê tiêu đề; phải dựa vào nội dung trong từng bài. +- Không lặp ý giữa các nguồn. +- Tối đa 6 gạch đầu dòng, mỗi dòng 1 câu rõ ràng. +- Nếu các nguồn có góc nhìn khác nhau, gộp lại thành ý tổng hợp. +- Cuối cùng thêm dòng: Nguồn tham khảo: tên website. + +Nội dung nguồn: +{ctx[:16000]}""" + text = await prev.base.qwen_generate(prompt, image_url=image or None, max_tokens=1100) + text = prev._postprocess_ai_text(text, max_units=7) + if 'Nguồn tham khảo:' not in text: + text += '\n\n' + prev._source_line(sources) + post = base.make_post('Tổng hợp: ' + topic, text, image or base.pollinations_image_url(topic), '', 'topic_aggregate', sources=sources[:5]) + posts = base._load_ai_wall(); posts.insert(0, post); base._save_ai_wall(posts) + return JSONResponse({'post': post, 'count_sources': len(sources)}) + + +@app.post('/api/ai/short/{post_id}') +async def ai_short_full(post_id: str, request: Request): + try: + body = await request.json() + except Exception: + body = {} + voice = str(body.get('voice','nu')).lower().strip() + emotion = str(body.get('emotion','neutral')).lower().strip() + speed = max(0.85, min(1.35, float(body.get('speed', 1.2) or 1.2))) + posts = base._load_ai_wall() + post = next((p for p in posts if str(p.get('id')) == str(post_id)), None) + if not post: + return JSONResponse({'error':'post not found'}, status_code=404) + os.makedirs(base.SHORTS_DIR, exist_ok=True) + suffix = f"_{voice}_{emotion}_{str(speed).replace('.', 'p')}_fullv2" + out_mp4 = os.path.join(base.SHORTS_DIR, base._safe_name(post_id + suffix) + '.mp4') + if os.path.exists(out_mp4): + post['video'] = '/api/ai/short-file/' + post_id + suffix + base._save_ai_wall(posts) + return JSONResponse({'video': post['video'], 'speed': speed, 'subtitles': True}) + work = os.path.join(base.SHORTS_DIR, base._safe_name(post_id + suffix)); os.makedirs(work, exist_ok=True) + img = os.path.join(work,'image.jpg'); frame = os.path.join(work,'frame.jpg'); audio = os.path.join(work,'voice.mp3'); audio_fast=os.path.join(work,'voice_fast.mp3'); srt=os.path.join(work,'subtitles.srt'); vtt=os.path.join(work,'subtitles.vtt') + try: + base._download_image(post.get('img'), post.get('title','AI news'), img) + prev._make_short_frame_full(post, img, frame) + script = tts_script_full(post, emotion) + edge_voice = {'nam':'vi-VN-NamMinhNeural','male':'vi-VN-NamMinhNeural','nu':'vi-VN-HoaiMyNeural','female':'vi-VN-HoaiMyNeural','mien-nam':'vi-VN-HoaiMyNeural'}.get(voice,'vi-VN-HoaiMyNeural') + used_edge = False + try: + subprocess.run(['python','-m','edge_tts','--voice',edge_voice,'--text',script,'--write-media',audio,'--write-subtitles',vtt], check=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=260) + used_edge = True + except Exception: + tld = 'com.vn' if voice in ('nu','female','mien-nam') else 'com' + try: + base.gTTS(script, lang='vi', tld=tld, slow=False).save(audio) + except TypeError: + base.gTTS(script, lang='vi', slow=False).save(audio) + subprocess.run(['ffmpeg','-y','-i',audio,'-filter:a',f'atempo={speed}','-vn',audio_fast], check=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=220) + duration = 45.0 + try: + pr = subprocess.run(['ffprobe','-v','error','-show_entries','format=duration','-of','default=noprint_wrappers=1:no_key=1',audio_fast], stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=20) + duration = float((pr.stdout or b'45').decode().strip() or 45) + except Exception: + pass + if used_edge and os.path.exists(vtt): + ok = convert_vtt_to_scaled_srt(vtt, srt, speed=speed) + if not ok: + write_weighted_srt(script, srt, duration) + else: + write_weighted_srt(script, srt, duration) + vf = "scale=1080:1920,subtitles='{}':force_style='FontName=DejaVu Sans,FontSize=16,PrimaryColour=&H00FFFFFF,OutlineColour=&HAA000000,BorderStyle=1,Outline=1.5,Shadow=0,Alignment=2,MarginV=42'".format(srt.replace("'", "\\'")) + cmd = ['ffmpeg','-y','-loop','1','-i',frame,'-i',audio_fast,'-shortest','-c:v','libx264','-tune','stillimage','-pix_fmt','yuv420p','-c:a','aac','-b:a','128k','-vf',vf,out_mp4] + subprocess.run(cmd, check=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=420) + post['video'] = '/api/ai/short-file/' + post_id + suffix + post['short_voice'] = voice; post['short_emotion'] = emotion; post['short_speed'] = speed; post['short_subtitles'] = True + base._save_ai_wall(posts) + return JSONResponse({'video': post['video'], 'voice': voice, 'emotion': emotion, 'speed': speed, 'subtitles': True, 'duration': duration}) + except Exception as e: + return JSONResponse({'error':'Không tạo được shorts: '+str(e)[:180]}, status_code=500) + + +@app.get('/api/ai/short-file/{file_id}') +def ai_short_file_full(file_id: str): + path = os.path.join(base.SHORTS_DIR, base._safe_name(file_id) + '.mp4') + if not os.path.exists(path): + return JSONResponse({'error':'not found'}, status_code=404) + return FileResponse(path, media_type='video/mp4', filename=f'vnews-ai-{file_id}.mp4') + + +app.router.routes = [r for r in app.router.routes if not (getattr(r,'path',None)=='/' and 'GET' in getattr(r,'methods',set()))] + +@app.get('/') +async def index_fix2(): + with open('/app/static/index.html','r',encoding='utf-8') as f: + html = f.read() + inject = prev.PATCH_INJECT + r''' + +''' + return HTMLResponse(html.replace('', inject+'\n')) diff --git a/ai_patch.py b/ai_patch.py new file mode 100644 index 0000000000000000000000000000000000000000..41aeba3d810744429c14e473c32c3a9fb8e3a605 --- /dev/null +++ b/ai_patch.py @@ -0,0 +1,917 @@ +import os +import re +import time +import random +import json +import html as html_lib +import subprocess +import requests +import hashlib +import ai_ext as base +from ai_ext import app +from fastapi import Request +from fastapi.responses import JSONResponse, HTMLResponse, FileResponse +from bs4 import BeautifulSoup +from urllib.parse import quote_plus + +try: + from PIL import Image, ImageDraw, ImageFont +except Exception: + Image = ImageDraw = ImageFont = None + + +def _clean(s): + s = html_lib.unescape(s or "") + s = re.sub(r"[ \t]+", " ", s) + s = re.sub(r"\n{3,}", "\n\n", s) + return s.strip() + + +def _norm(s): + s = s.lower() + s = re.sub(r"[^\wÀ-ỹ\s]", " ", s) + s = re.sub(r"\s+", " ", s).strip() + return s + + +def _similar(a, b): + ta = set(_norm(a).split()) + tb = set(_norm(b).split()) + if not ta or not tb: + return False + return len(ta & tb) / max(1, min(len(ta), len(tb))) >= 0.72 + + +def _dedupe_units(units, max_units=25): + """Deduplicate units - only skip exact matches to ensure all bullet points are read.""" + out, seen = [], set() + for u in units: + u = _clean(re.sub(r"^[-•*\d\.\)\s]+", "", u)) + if len(u) < 18: + continue + nu = _norm(u) + # Only skip exact matches, NOT similar content (to avoid skipping valid bullet points) + if nu in seen: + continue + seen.add(nu) + out.append(u) + if len(out) >= max_units: + break + return out + + +def _postprocess_ai_text(text, max_units=20): + text = _clean(text) + if not text: + return text + drop_prefixes = ( + "dưới đây", "sau đây", "bài viết", "tôi sẽ", "mình sẽ", + "tóm tắt bài", "tiêu đề:", "sapo:", "nội dung:", "kết luận:" + ) + raw_lines = [] + for line in re.split(r"\n+", text): + line = _clean(line) + if not line: + continue + low = line.lower().strip() + if any(low.startswith(p) and len(line) < 80 for p in drop_prefixes): + continue + raw_lines.append(line) + units = [] + for line in raw_lines: + # KEEP FULL bullet point - don't truncate or split into segments + if len(line) >= 18: + units.append(_clean(re.sub(r"^[-•*\d\.\)\s]+", "", line))) + units = _dedupe_units(units, max_units=max_units) + if not units: + return text[:900] + title = "" + if raw_lines and len(raw_lines[0]) <= 90 and not raw_lines[0].startswith(("-", "•", "*")): + title = raw_lines[0] + units = [u for u in units if not _similar(u, title)] + body = "\n".join("• " + u for u in units[:max_units]) + return (title + "\n\n" + body).strip() if title else body + + +def _fallback_summary_from_prompt(prompt, max_units=6): + text = prompt or "" + for marker in ["Nội dung nguồn:", "Nội dung bài:", "Nội dung gốc:", "Nội dung:", "Nguồn/bối cảnh internet:"]: + if marker in text: + text = text.split(marker, 1)[1] + break + text = re.sub(r"https?://\S+", "", text) + text = re.sub(r"\s+", " ", text).strip() + sentences = re.split(r"(?<=[\.\!\?])\s+(?=[A-ZÀ-Ỹ0-9])", text) + candidates = [] + for s in sentences: + s = _clean(s) + if 45 <= len(s) <= 260: + candidates.append(s) + units = _dedupe_units(candidates, max_units=max_units) + if units: + return "\n".join("• " + u for u in units) + if text: + return "• " + text[:700].rsplit(" ", 1)[0] + return "• Không có đủ nội dung nguồn để tóm tắt." + + +def _source_line(sources): + names = [] + for s in (sources or [])[:5]: + via = s.get("via") or base._domain(s.get("url", "")) or s.get("title", "") + if via and via not in names: + names.append(via) + return "Nguồn tham khảo: " + ", ".join(names[:5]) if names else "Nguồn tham khảo: tổng hợp internet" + + +def _make_summary_prompt(title, raw, source_hint=""): + return f"""Bạn là biên tập viên tóm tắt tin tức tiếng Việt. + +NHIỆM VỤ BẮT BUỘC: +- Chỉ TÓM TẮT nội dung chính, KHÔNG viết lại toàn bộ bài. +- Không lặp lại cùng một ý, cùng một câu, cùng một chi tiết. +- Không thêm thông tin ngoài nguồn. +- Tối đa 5 gạch đầu dòng, mỗi gạch đầu dòng 1 câu ngắn. +- Nếu bài có số liệu/nhân vật/thời điểm quan trọng thì giữ lại. +- Không viết phần mở bài dài, không viết văn kể lại. + +Tiêu đề nguồn: {title} +Nguồn: {source_hint} + +Nội dung nguồn: +{raw[:14000]} +""" + + +def _direct_news_rss(topic, limit=10): + out = [] + try: + url = "https://news.google.com/rss/search?q=" + quote_plus(topic) + "&hl=vi&gl=VN&ceid=VN:vi" + r = requests.get(url, headers=base.HEADERS, timeout=15) + r.encoding = "utf-8" + soup = BeautifulSoup(r.text, "xml") + for it in soup.find_all("item")[:limit]: + title = it.find("title").get_text(" ", strip=True) if it.find("title") else "" + link = it.find("link").get_text(strip=True) if it.find("link") else "" + src = it.find("source").get_text(" ", strip=True) if it.find("source") else base._domain(link) + if title and link: + out.append({"title": title, "url": link, "via": src, "excerpt": title}) + except Exception: + pass + return out + + +def _topic_source_articles(topic, limit=5): + """Return actual scraped article bodies for a topic. Each source becomes one Wall AI post.""" + try: + _ctx, sources = base.web_context(topic, limit=limit) + except Exception: + sources = [] + if not sources: + sources = _direct_news_rss(topic, limit=10) + out, seen = [], set() + for s in (sources or [])[:limit * 3]: + url = s.get("url") or "" + if not url.startswith("http") or url in seen: + continue + seen.add(url) + try: + page = base.scrape_any_url(url) + raw = (page.get("summary", "") + "\n" + page.get("text", "")).strip() + if len(raw) < 180: + continue + title = page.get("title") or s.get("title") or url + via = page.get("via") or s.get("via") or base._domain(url) + out.append({ + "title": title, + "url": url, + "raw": raw, + "image": page.get("image") or "", + "via": via, + "source": {"title": title, "url": url, "excerpt": raw[:700], "via": via} + }) + if len(out) >= limit: + break + except Exception: + continue + if not out: + for s in (sources or _direct_news_rss(topic, 6))[:limit]: + title = s.get("title") or topic + excerpt = s.get("excerpt") or s.get("description") or s.get("content") or title + url = s.get("url", "") + via = s.get("via") or base._domain(url) + out.append({ + "title": title, + "url": url, + "raw": excerpt, + "image": base.pollinations_image_url(title), + "via": via, + "source": {"title": title, "url": url, "excerpt": excerpt[:700], "via": via} + }) + return out[:limit] + + +async def qwen_generate_resilient(prompt: str, image_url=None, max_tokens: int = 1200): + errors = [] + token = base._hf_token() + try: + original = getattr(base, "_original_qwen_generate", None) + if original: + txt = await original(prompt, image_url=image_url, max_tokens=max_tokens) + if txt: + base.LAST_QWEN_ERROR = "" + return txt + if getattr(base, "LAST_QWEN_ERROR", ""): + errors.append("sdk: " + str(base.LAST_QWEN_ERROR)[:260]) + except Exception as e: + errors.append(f"sdk: {type(e).__name__}: {str(e)[:260]}") + if token: + models = [] + for m in [ + os.getenv("QWEN_VL_MODEL", ""), + "Qwen/Qwen2.5-VL-7B-Instruct", + "Qwen/Qwen2.5-VL-3B-Instruct", + "Qwen/Qwen2.5-7B-Instruct", + "Qwen/Qwen2.5-3B-Instruct", + "Qwen/Qwen2.5-1.5B-Instruct", + ]: + if m and m not in models: + models.append(m) + headers = {"Authorization": "Bearer " + token, "Content-Type": "application/json"} + for model in models: + try: + is_vl = "VL" in model and bool(image_url) + user_content = ([{"type": "image_url", "image_url": {"url": image_url}}, {"type": "text", "text": prompt}] if is_vl else prompt) + payload = { + "model": model, + "messages": [ + {"role": "system", "content": "Bạn là biên tập viên AI tiếng Việt. Chỉ tóm tắt súc tích nội dung nguồn, không viết lại toàn bài, không lặp ý, không bịa chi tiết."}, + {"role": "user", "content": user_content}, + ], + "max_tokens": min(int(max_tokens or 900), 1400), + "temperature": 0.35, + "top_p": 0.85, + } + r = requests.post("https://router.huggingface.co/v1/chat/completions", headers=headers, json=payload, timeout=95) + if r.status_code >= 300: + errors.append(f"{model}: HTTP {r.status_code} {r.text[:180]}") + continue + j = r.json() + txt = (j.get("choices", [{}])[0].get("message", {}).get("content") or "").strip() + if txt: + base.LAST_QWEN_ERROR = "" + return txt + errors.append(f"{model}: empty response") + except Exception as e: + errors.append(f"{model}: {type(e).__name__}: {str(e)[:220]}") + else: + errors.append("missing HF_TOKEN") + base.LAST_QWEN_ERROR = " | ".join(errors[-6:]) or "Qwen unavailable; used extractive fallback" + print("[qwen resilient fallback]", base.LAST_QWEN_ERROR) + return _fallback_summary_from_prompt(prompt, max_units=12) + + +if not hasattr(base, "_original_qwen_generate"): + base._original_qwen_generate = base.qwen_generate +base.qwen_generate = qwen_generate_resilient + + +@app.get('/api/wall') +def compat_wall(): + return JSONResponse({'posts': base._load_ai_wall()[:80]}) + + +_PATCHED_PATHS = { + ('/api/topic_post', 'POST'), + ('/api/url_wall', 'POST'), + ('/api/rewrite_share', 'POST'), + ('/api/ai/short/{post_id}', 'POST'), +} +app.router.routes = [ + r for r in app.router.routes + if not any(getattr(r, 'path', None) == p and m in getattr(r, 'methods', set()) for p, m in _PATCHED_PATHS) +] + + +@app.post('/api/topic_post') +async def compat_topic_post(request: Request): + body = await request.json() + topic = base._clean_text(body.get('topic', '')) + if not topic: + return JSONResponse({'error': 'missing topic'}, status_code=400) + articles = _topic_source_articles(topic, limit=4) + if not articles: + return JSONResponse({'error': 'Không lấy được bài viết nguồn cho chủ đề này.'}, status_code=422) + new_posts = [] + posts = base._load_ai_wall() + for art in articles: + prompt = f"""Tóm tắt RIÊNG bài viết nguồn sau để đăng Tường AI. + +Chủ đề lọc: {topic} +Tiêu đề bài nguồn: {art['title']} +Nguồn: {art['via']} + +Yêu cầu bắt buộc: +- Tóm tắt nội dung trong BÀI VIẾT này, không chỉ tiêu đề. +- Không trộn với bài khác. +- Không viết lại toàn bộ bài. +- Không lặp ý. +- 4-6 gạch đầu dòng, mỗi dòng 1 câu rõ ràng. +- Giữ số liệu/nhân vật/thời điểm quan trọng nếu có. + +Nội dung bài: +{art['raw'][:14000]}""" + text = await base.qwen_generate(prompt, image_url=art.get('image') or None, max_tokens=1500) + text = _postprocess_ai_text(text, max_units=20) + src = [art['source']] + if 'Nguồn tham khảo:' not in text: + text += "\n\n" + _source_line(src) + post = base.make_post(art['title'], text, art.get('image') or base.pollinations_image_url(art['title']), art.get('url') or '', 'topic_article', sources=src) + + # Generate slides for this post so they persist after page reload + try: + page_data = _scrape_article_images(art.get('url', '')) + if page_data and page_data.get('paragraphs'): + key_points = _extract_key_points_for_slides(page_data['paragraphs'], max_points=12) + if key_points: + relevant_imgs = page_data.get('images', []) + if not relevant_imgs and page_data.get('og_img'): + relevant_imgs = [page_data['og_img']] + slides = [] + for i, point in enumerate(key_points): + img = relevant_imgs[i] if i < len(relevant_imgs) else (relevant_imgs[-1] if relevant_imgs else '') + slides.append({'text': point, 'image': img, 'index': i + 1}) + post['slides'] = slides + except Exception: + pass + + new_posts.append(post) + posts = new_posts + posts + base._save_ai_wall(posts) + return JSONResponse({'post': new_posts[0], 'posts': new_posts, 'count': len(new_posts)}) + + +@app.post('/api/url_wall') +async def compat_url_wall(request: Request): + body = await request.json() + url = base._clean_text(body.get('url', '')) + if not url.startswith('http'): + return JSONResponse({'error': 'missing url'}, status_code=400) + try: + data = base.scrape_any_url(url) + except Exception as e: + return JSONResponse({'error': 'Không scrape được URL: ' + str(e)[:180]}, status_code=422) + raw = (data.get('summary', '') + '\n' + data.get('text', '')).strip() + if len(raw) < 120: + return JSONResponse({'error': 'URL không có đủ nội dung để tóm tắt'}, status_code=422) + prompt = _make_summary_prompt(data.get('title', ''), raw, data.get('via', '') or base._domain(url)) + text = await base.qwen_generate(prompt, image_url=data.get('image') or None, max_tokens=1500) + text = _postprocess_ai_text(text, max_units=20) + src = [{'title': data.get('title'), 'url': url, 'excerpt': raw[:500], 'via': data.get('via') or base._domain(url)}] + if 'Nguồn tham khảo:' not in text: + text += "\n\n" + _source_line(src) + post = base.make_post(data.get('title') or 'Bài viết', text, data.get('image') or '', url, 'url', sources=src) + + # Generate slides so they persist after page reload + slides = [] + try: + page_data = _scrape_article_images(url) + if page_data and page_data.get('paragraphs'): + key_points = _extract_key_points_for_slides(page_data['paragraphs'], max_points=12) + if key_points: + relevant_imgs = page_data.get('images', []) + if not relevant_imgs and page_data.get('og_img'): + relevant_imgs = [page_data['og_img']] + for i, point in enumerate(key_points): + img = relevant_imgs[i] if i < len(relevant_imgs) else (relevant_imgs[-1] if relevant_imgs else '') + slides.append({'text': point, 'image': img, 'index': i + 1}) + except Exception: + pass + post['slides'] = slides + + posts = base._load_ai_wall(); posts.insert(0, post); base._save_ai_wall(posts) + return JSONResponse({'post': post, 'slides': slides}) + + +def _is_relevant_image(img_url, title, text): + """Check if an image is relevant to the article content.""" + if not img_url: + return False + skip_patterns = ['pixel', 'analytics', 'tracking', '1x1.gif', 'spacer.gif', + 'logo', 'icon', 'avatar', 'emoji', 'smiley', 'sprite', + 'advertisement', 'ad-banner', 'sponsored', 'banner-ads'] + img_lower = img_url.lower() + for p in skip_patterns: + if p in img_lower: + return False + if not any(img_lower.endswith(ext) for ext in ['.jpg', '.jpeg', '.png', '.webp', '.gif']): + return False + return True + + +def _filter_relevant_images(images, title, text, max_images=8): + """Filter and rank images by relevance to article content.""" + if not images: + return [] + seen = set() + relevant = [] + for img in images: + if img in seen: + continue + seen.add(img) + if _is_relevant_image(img, title, text): + relevant.append(img) + return relevant[:max_images] + + +def _extract_key_points_for_slides(paragraphs, max_points=12): + """Extract key points from paragraphs for slides - extracts ALL sentences, not just first one.""" + points = [] + for p in paragraphs: + if len(points) >= max_points: + break + p = _clean(p) + if not p: + continue + # Split paragraph into sentences using Vietnamese + English punctuation - GET ALL SENTENCES + sentences = re.split(r'(?<=[.!?])\s+(?=[A-ZÀ-Ỹ0-9])', p) + sentences = [s.strip() for s in sentences if s.strip()] + + for sentence in sentences: + if len(points) >= max_points: + break + sentence = _clean(sentence) + if len(sentence) < 30: + continue + if any(sentence[:60] in existing for existing in points): + continue + if not sentence.endswith(('.', '!', '?')): + sentence = sentence + '.' + points.append(sentence) + return points + + +def _scrape_article_images(url): + """Scrape article page and return only relevant images.""" + try: + headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36", + "Accept-Language": "vi-VN,vi;q=0.9,en;q=0.8"} + r = requests.get(url, headers=headers, timeout=15, allow_redirects=True) + r.encoding = 'utf-8' + soup = BeautifulSoup(r.text, 'lxml') + for tag in soup.find_all(['script', 'style', 'nav', 'footer', 'aside', 'form']): + tag.decompose() + h1 = soup.find('h1') + ogt = soup.find('meta', property='og:title') + title = (h1.get_text(strip=True) if h1 else '') or (ogt.get('content', '') if ogt else '') + ogi = soup.find('meta', property='og:image') + og_img = ogi.get('content', '') if ogi else '' + if og_img and og_img.startswith('//'): + og_img = 'https:' + og_img + block = None + for sel in ['article', '.singular-content', '.detail-content', '.fck_detail', '.content-detail', '.knc-content', 'main', '.cms-body', '.article__body']: + el = soup.select_one(sel) + if el and len(el.find_all('p')) >= 2: + block = el + break + if not block: + block = soup.body or soup + paragraphs = [] + all_images = [] + seen_imgs = set() + if og_img and og_img not in seen_imgs: + all_images.append(og_img) + seen_imgs.add(og_img) + for el in block.find_all(['p', 'h2', 'h3', 'figure', 'img'], recursive=True): + if el.name == 'p': + t = _clean(el.get_text(strip=True)) + if t and len(t) > 40: + paragraphs.append(t) + elif el.name in ('figure', 'img'): + im = el if el.name == 'img' else el.find('img') + if im: + src = im.get('data-src') or im.get('src') or im.get('data-original') or '' + if src and 'base64' not in src: + if src.startswith('//'): + src = 'https:' + src + if src not in seen_imgs: + all_images.append(src) + seen_imgs.add(src) + relevant_images = _filter_relevant_images(all_images, title, ' '.join(paragraphs[:5])) + return {'title': _clean(title), 'paragraphs': paragraphs, 'images': relevant_images, 'og_img': og_img} + except Exception: + return None + + +@app.post('/api/rewrite_share') +async def compat_rewrite_share(request: Request): + body = await request.json() + url = base._clean_text(body.get('url', '')) + if not url.startswith('http'): + return JSONResponse({'error': 'missing url'}, status_code=400) + try: + data = base.scrape_any_url(url) + except Exception as e: + return JSONResponse({'error': 'Không đọc được bài viết: ' + str(e)[:180]}, status_code=422) + raw = (data.get('summary', '') + '\n' + data.get('text', '')).strip() + if len(raw) < 120: + return JSONResponse({'error': 'Bài viết không đủ nội dung để tóm tắt'}, status_code=422) + prompt = _make_summary_prompt(data.get('title', ''), raw, data.get('via', '') or base._domain(url)) + text = await base.qwen_generate(prompt, image_url=data.get('image') or None, max_tokens=1500) + text = _postprocess_ai_text(text, max_units=20) + src = [{'title': data.get('title'), 'url': url, 'excerpt': raw[:500], 'via': data.get('via') or base._domain(url)}] + if 'Nguồn tham khảo:' not in text: + text += "\n\n" + _source_line(src) + post = base.make_post(data.get('title') or 'Bài viết', text, data.get('image') or '', url, 'summary', sources=src) + + # Generate slides with relevant images only + slides = [] + page_data = _scrape_article_images(url) + if page_data and page_data.get('paragraphs'): + key_points = _extract_key_points_for_slides(page_data['paragraphs'], max_points=12) + if key_points: + relevant_imgs = page_data.get('images', []) + if not relevant_imgs and page_data.get('og_img'): + relevant_imgs = [page_data['og_img']] + for i, point in enumerate(key_points): + img = relevant_imgs[i] if i < len(relevant_imgs) else (relevant_imgs[-1] if relevant_imgs else '') + slides.append({'text': point, 'image': img, 'index': i + 1}) + + # FIX: Save slides into post so they persist after page reload + post['slides'] = slides + posts = base._load_ai_wall(); posts.insert(0, post); base._save_ai_wall(posts) + + return JSONResponse({'post': post, 'slides': slides}) + + +def _emotion_script(text, emotion): + """Prepend emotion-appropriate prefix to text based on emotion type. + + NOTE: Prefix is NOT added to avoid cluttering Short AI speech. + The emotion is still used for voice selection but content is read cleanly. + """ + text = _clean(text) + # REMOVED: No prefix added to keep content clean and natural + return text + + +def _tts_script_smart(post, emotion): + raw = base._short_script(post) if hasattr(base, '_short_script') else _clean(post.get('text', '') or post.get('title', '')) + raw = re.sub(r"^[•\-\*]\s*", "", raw, flags=re.M) + raw = re.sub(r"\s*\n\s*", ". ", raw) + raw = re.sub(r"([\.\!\?])\s*", r"\1\n", raw) + raw = re.sub(r"\n{2,}", "\n", raw).strip() + # REMOVED: _emotion_script call - read content cleanly without prefix + # INCREASED to 3000 to read full content of all bullet points + if len(raw) > 3000: + raw = raw[:3000] + cut = max(raw.rfind("."), raw.rfind("!"), raw.rfind("?")) + if cut > 700: + raw = raw[:cut + 1] + return raw + + +def _split_subtitle_sentences(script): + parts = [] + for line in script.splitlines(): + line = _clean(line) + if not line: + continue + for s in re.split(r"(?<=[\.\!\?])\s+", line): + s = _clean(s) + if 8 <= len(s) <= 140: + parts.append(s) + return parts[:12] + + +def _srt_time(sec): + ms = int((sec - int(sec)) * 1000) + sec = int(sec) + h = sec // 3600 + m = (sec % 3600) // 60 + s = sec % 60 + return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}" + + +def _write_srt(script, path, total_duration=30): + subs = _split_subtitle_sentences(script) + if not subs: + subs = [script[:120]] + dur = max(2.2, min(5.0, total_duration / max(1, len(subs)))) + cur = 0.3 + with open(path, 'w', encoding='utf-8') as f: + for i, s in enumerate(subs, 1): + start = cur + end = cur + dur + cur = end + 0.15 + f.write(f"{i}\n{_srt_time(start)} --> {_srt_time(end)}\n{s}\n\n") + + +def _wrap_text_px(draw, text, font, max_width, max_lines): + words = _clean(text).split() + lines, cur = [], "" + for w in words: + test = (cur + " " + w).strip() + try: + width = draw.textbbox((0, 0), test, font=font)[2] + except Exception: + width = len(test) * 20 + if width <= max_width: + cur = test + else: + if cur: + lines.append(cur) + cur = w + if len(lines) >= max_lines: + break + if cur and len(lines) < max_lines: + lines.append(cur) + return lines + + +def _make_short_frame_full(post, img_path, out_path): + if Image is None: + return base._make_short_frame(post, img_path, out_path) + W, H = 1080, 1920 + bg = Image.new("RGB", (W, H), (14, 14, 14)) + try: + im = Image.open(img_path).convert("RGB") + target = (1080, 760) + im_ratio = im.width / im.height + target_ratio = target[0] / target[1] + if im_ratio > target_ratio: + new_h = target[1] + new_w = int(new_h * im_ratio) + else: + new_w = target[0] + new_h = int(new_w / im_ratio) + im = im.resize((new_w, new_h)) + left = (new_w - target[0]) // 2 + top = (new_h - target[1]) // 2 + im = im.crop((left, top, left + target[0], top + target[1])) + bg.paste(im, (0, 0)) + except Exception: + pass + draw = ImageDraw.Draw(bg) + try: + font_title = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf", 54) + font_body = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf", 38) + font_label = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf", 30) + except Exception: + font_title = font_body = font_label = None + draw.rectangle((0, 720, W, H), fill=(14, 14, 14)) + margin = 48 + maxw = W - margin * 2 + draw.text((margin, 770), "VNEWS · Tường AI", fill=(92, 184, 122), font=font_label) + y = 830 + for ln in _wrap_text_px(draw, post.get("title", ""), font_title, maxw, 4): + draw.text((margin, y), ln, fill=(255, 255, 255), font=font_title) + y += 66 + y += 18 + text = post.get("text", "") + text = re.sub(r"Nguồn tham khảo:.*", "", text, flags=re.S).strip() + body_lines = _wrap_text_px(draw, text, font_body, maxw, 14) + for ln in body_lines: + draw.text((margin, y), ln, fill=(220, 220, 220), font=font_body) + y += 50 + if y > 1640: + break + bg.save(out_path, quality=92) + + + + +def _summary_segments_from_post(post, max_segments=25): + raw = _clean(post.get('text') or post.get('title') or '') + raw = re.sub(r'^Bản tin AI viết lại:\s*', '', raw, flags=re.I) + raw = re.sub(r'Nguồn tham khảo:.*$', '', raw, flags=re.I|re.S).strip() + lines=[] + for ln in raw.splitlines(): + ln=_clean(re.sub(r'^[•\-\*\d\.\)\s]+','',ln)) + if not ln: continue + low=ln.lower() + if low.startswith(('điểm chính','tiêu đề','sapo','nguồn tham khảo')): continue + if len(ln)>=18: lines.append(ln) + if len(lines)<3: + lines=[] + for s in re.split(r'(?<=[\.\!\?])\s+', raw): + s=_clean(s) + if len(s)>=25: lines.append(s) + segs=_dedupe_units(lines, max_units=max_segments) + return segs[:max_segments] if segs else [post.get('title','Bản tin VNEWS')] + + +def _make_scene_frame(post, segment, idx, total, img_path, out_path, emotion='neutral'): + if Image is None: + return _make_short_frame_full(post, img_path, out_path) + W,H=1080,1920 + bg=Image.new('RGB',(W,H),(10,10,10)) + try: + im=Image.open(img_path).convert('RGB') + ratio=im.width/max(1,im.height); target=W/H + if ratio>target: + nh=H; nw=int(nh*ratio) + else: + nw=W; nh=int(nw/ratio) + cover=im.resize((nw,nh)); left=(nw-W)//2; top=(nh-H)//2 + cover=cover.crop((left,top,left+W,top+H)) + bg.paste(cover,(0,0)) + bg=Image.blend(bg, Image.new('RGB',(W,H),(0,0,0)), 0.50) + hero_h=720; target=W/hero_h + if ratio>target: + nh=hero_h; nw=int(nh*ratio) + else: + nw=W; nh=int(nw/ratio) + hero=im.resize((nw,nh)); left=(nw-W)//2; top=(nh-hero_h)//2 + hero=hero.crop((left,top,left+W,top+hero_h)) + bg.paste(hero,(0,0)) + except Exception: + pass + draw=ImageDraw.Draw(bg) + try: + font_brand=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',34) + font_small=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf',28) + font_seg=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',58) + font_title=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf',34) + except Exception: + font_brand=font_small=font_seg=font_title=None + draw.rectangle((0,680,W,H), fill=(12,12,12)) + dot_x=48; dot_y=742 + for i in range(total): + fill=(92,184,122) if i==idx else (70,70,70) + draw.rounded_rectangle((dot_x+i*38,dot_y,dot_x+i*38+24,dot_y+10), radius=5, fill=fill) + draw.text((48,780),'VNEWS AI SHORT',fill=(110,231,143),font=font_brand) + draw.rounded_rectangle((48,834,260,880), radius=20, fill=(28,70,45)) + draw.text((66,842),f'Đoạn {idx+1}/{total}',fill=(235,235,235),font=font_small) + y=940; maxw=W-96 + # INCREASED from 12 to 18 for full content display - each key point can span multiple lines + for ln in _wrap_text_px(draw, segment, font_seg, maxw, 18): + draw.text((48,y),ln,fill=(255,255,255),font=font_seg) + y+=74 + if y>1500: break + y2=1640 + draw.line((48,y2-22,W-48,y2-22),fill=(70,70,70),width=2) + for ln in _wrap_text_px(draw, post.get('title',''), font_title, maxw, 3): + draw.text((48,y2),ln,fill=(220,220,220),font=font_title) + y2+=46 + bg.save(out_path, quality=92) + + +def _estimate_audio_duration(path, fallback=15.0): + """Estimate audio duration with 15s minimum per segment for complete bullet reading.""" + try: + pr=subprocess.run(['ffprobe','-v','error','-show_entries','format=duration','-of','default=noprint_wrappers=1:no_key=1',path], stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=20) + return max(12.0, float((pr.stdout or b'').decode().strip() or fallback)) + except Exception: + return fallback + + +@app.post('/api/ai/short/{post_id}') +async def patched_ai_short(post_id: str, request: Request): + try: + body = await request.json() + except Exception: + body = {} + voice = str(body.get('voice', 'nu')).strip().lower() + emotion = str(body.get('emotion', 'neutral')).strip().lower() + speed = float(body.get('speed', 1.0) or 1.0) + speed = max(0.85, min(1.35, speed)) + + posts = base._load_ai_wall() + post = next((p for p in posts if str(p.get('id')) == str(post_id)), None) + if not post: + return JSONResponse({'error': 'post not found'}, status_code=404) + + segments = _summary_segments_from_post(post, max_segments=25) + seg_hash = hashlib.md5(('|'.join(segments)+voice+emotion+str(speed)).encode('utf-8')).hexdigest()[:8] + os.makedirs(base.SHORTS_DIR, exist_ok=True) + suffix = f"_{voice}_{emotion}_{str(speed).replace('.', 'p')}_{seg_hash}_scenes_nosub" + out_mp4 = os.path.join(base.SHORTS_DIR, base._safe_name(post_id + suffix) + '.mp4') + if os.path.exists(out_mp4): + post['video'] = '/api/ai/short-file/' + post_id + suffix + post['short_voice'] = voice + post['short_emotion'] = emotion + post['short_speed'] = speed + post['short_segments'] = segments + post['short_subtitles'] = False + base._save_ai_wall(posts) + return JSONResponse({'video': post['video'], 'voice': voice, 'emotion': emotion, 'speed': speed, 'subtitles': False, 'segments': segments}) + if base.gTTS is None: + return JSONResponse({'error': 'gTTS chưa sẵn sàng'}, status_code=503) + + work = os.path.join(base.SHORTS_DIR, base._safe_name(post_id + suffix)) + os.makedirs(work, exist_ok=True) + img = os.path.join(work, 'image.jpg') + try: + base._download_image(post.get('img'), post.get('title', 'AI news'), img) + edge_voice = { + # Vietnamese + 'vi-vn-hoaimyneural': 'vi-VN-HoaiMyNeural', + 'vi-vn-namminhneural': 'vi-VN-NamMinhNeural', + 'hoaimy': 'vi-VN-HoaiMyNeural', + 'namminh': 'vi-VN-NamMinhNeural', + 'nam': 'vi-VN-NamMinhNeural', + 'male': 'vi-VN-NamMinhNeural', + 'nu': 'vi-VN-HoaiMyNeural', + 'female': 'vi-VN-HoaiMyNeural', + 'mien-nam': 'vi-VN-HoaiMyNeural', + # English - Multilingual + 'en-us-andrewmultilingualneural': 'en-US-AndrewMultilingualNeural', + 'en-au-williammultilingualneural': 'en-AU-WilliamMultilingualNeural', + 'andrew': 'en-US-AndrewMultilingualNeural', + 'en_andrew': 'en-US-AndrewMultilingualNeural', + 'jenny': 'en-US-AndrewMultilingualNeural', + 'en_jenny': 'en-US-AndrewMultilingualNeural', + # Portuguese - Multilingual (ONLY Thalita) + 'pt-br-thalitamultilingualneural': 'pt-BR-ThalitaMultilingualNeural', + 'thalita': 'pt-BR-ThalitaMultilingualNeural', + 'pt_thalita': 'pt-BR-ThalitaMultilingualNeural', + 'pt_br_thalita': 'pt-BR-ThalitaMultilingualNeural', + 'pt': 'pt-BR-ThalitaMultilingualNeural', + 'pt_francisco': 'pt-BR-ThalitaMultilingualNeural', + # French - Multilingual + 'fr-fr-viviennemultilingualneural': 'fr-FR-VivienneMultilingualNeural', + 'fr-fr-remymultilingualneural': 'fr-FR-RemyMultilingualNeural', + 'denise': 'fr-FR-VivienneMultilingualNeural', + 'fr': 'fr-FR-VivienneMultilingualNeural', + 'fr_denise': 'fr-FR-VivienneMultilingualNeural', + # German - Multilingual + 'de-de-seraphinamultilingualneural': 'de-DE-SeraphinaMultilingualNeural', + 'de-de-florianmultilingualneural': 'de-DE-FlorianMultilingualNeural', + 'katja': 'de-DE-SeraphinaMultilingualNeural', + 'de': 'de-DE-SeraphinaMultilingualNeural', + 'de_katja': 'de-DE-SeraphinaMultilingualNeural', + # Korean - Multilingual (Hyunsu, NOT SunHee) + 'ko-kr-hyusumultilingualneural': 'ko-KR-HyunsuMultilingualNeural', + 'ko-kr-hyunsuneural': 'ko-KR-HyunsuMultilingualNeural', + 'sunhee': 'ko-KR-HyunsuMultilingualNeural', + 'ko': 'ko-KR-HyunsuMultilingualNeural', + 'ko_sunhee': 'ko-KR-HyunsuMultilingualNeural', + # Italian - Multilingual + 'it-it-giuseppemultilingualneural': 'it-IT-GiuseppeMultilingualNeural', + # Spanish (keep for backward compat) + 'ela': 'en-US-AndrewMultilingualNeural', + 'es_ela': 'en-US-AndrewMultilingualNeural', + 'es': 'en-US-AndrewMultilingualNeural', + 'es_carlos': 'en-US-AndrewMultilingualNeural', + # Japanese (keep for backward compat) + 'nanami': 'en-US-AndrewMultilingualNeural', + 'ja': 'en-US-AndrewMultilingualNeural', + 'ja_nanami': 'en-US-AndrewMultilingualNeural', + # Chinese (keep for backward compat) + 'xiaochen': 'en-US-AndrewMultilingualNeural', + 'zh': 'en-US-AndrewMultilingualNeural', + 'zh_xiaochen': 'en-US-AndrewMultilingualNeural', + }.get(voice, 'vi-VN-HoaiMyNeural') + part_files=[] + for idx, seg in enumerate(segments): + frame=os.path.join(work,f'frame_{idx:02d}.jpg') + aud=os.path.join(work,f'voice_{idx:02d}.mp3') + aud_fast=os.path.join(work,f'voice_{idx:02d}_fast.mp3') + part=os.path.join(work,f'part_{idx:02d}.mp4') + _make_scene_frame(post, seg, idx, len(segments), img, frame, emotion=emotion) + spoken=_emotion_script(seg, emotion) + try: + subprocess.run(['python','-m','edge_tts','--voice',edge_voice,'--text',spoken,'--write-media',aud], check=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=120) + except Exception: + tld='com.vn' if voice in ('nu','female','mien-nam','hoaimy') else 'com' + try: + base.gTTS(spoken, lang='vi', tld=tld, slow=False).save(aud) + except TypeError: + base.gTTS(spoken, lang='vi', slow=False).save(aud) + subprocess.run(['ffmpeg','-y','-i',aud,'-filter:a',f'atempo={speed}','-vn',aud_fast], check=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=90) + dur=_estimate_audio_duration(aud_fast, fallback=15.0)+0.35 + subprocess.run(['ffmpeg','-y','-loop','1','-t',str(dur),'-i',frame,'-i',aud_fast,'-shortest','-c:v','libx264','-tune','stillimage','-pix_fmt','yuv420p','-c:a','aac','-b:a','128k',part], check=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=150) + part_files.append(part) + concat=os.path.join(work,'concat.txt') + with open(concat,'w',encoding='utf-8') as f: + for p in part_files: + f.write("file '" + p.replace("'", "'\\''") + "'\n") + subprocess.run(['ffmpeg','-y','-f','concat','-safe','0','-i',concat,'-c','copy',out_mp4], check=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=180) + post['video'] = '/api/ai/short-file/' + post_id + suffix + post['short_voice'] = voice + post['short_emotion'] = emotion + post['short_speed'] = speed + post['short_segments'] = segments + post['short_subtitles'] = False + base._save_ai_wall(posts) + return JSONResponse({'video': post['video'], 'voice': voice, 'emotion': emotion, 'speed': speed, 'subtitles': False, 'segments': segments}) + except Exception as e: + return JSONResponse({'error': 'Không tạo được shorts: ' + str(e)[:220]}, status_code=500) + + +@app.get('/api/ai/short-file/{file_id}') +def patched_ai_short_file(file_id: str): + path = os.path.join(base.SHORTS_DIR, base._safe_name(file_id) + '.mp4') + if not os.path.exists(path): + return JSONResponse({'error': 'not found'}, status_code=404) + return FileResponse(path, media_type='video/mp4', filename=f'vnews-ai-{file_id}.mp4') + + +@app.get('/api/ai_shorts') +def api_ai_shorts(): + posts = [p for p in base._load_ai_wall() if p.get('video')] + return JSONResponse({'posts': posts[:80]}) + + +app.router.routes = [r for r in app.router.routes if not (getattr(r, 'path', None) == '/' and 'GET' in getattr(r, 'methods', set()))] diff --git a/ai_runtime.py b/ai_runtime.py new file mode 100644 index 0000000000000000000000000000000000000000..b644d317bcc589fefc470ceb871cab0b3149a717 --- /dev/null +++ b/ai_runtime.py @@ -0,0 +1,357 @@ +import os, re, subprocess, json, time, hashlib +import ai_patch as old +from ai_patch import app +import ai_ext as base +from fastapi import Request +from fastapi.responses import JSONResponse, HTMLResponse, FileResponse +try: + from PIL import Image, ImageDraw, ImageFont +except Exception: + Image = ImageDraw = ImageFont = None + + +def clean(s): + import html as html_lib + return re.sub(r"\s+", " ", html_lib.unescape(s or "")).strip() + + +def _domain(url): + try: + from urllib.parse import urlparse + return urlparse(url or '').netloc.replace('www.','') + except Exception: + return '' + + +def _strip_bullet_prefix(s): + # remove bullets, numbered prefixes, leading dots commonly produced by AI summaries + return clean(re.sub(r'^[\s•\-\*·▪▫●○\d\.\)\(]+', '', s or '')) + + +def source_line(sources): + names=[] + for s in (sources or [])[:5]: + via=s.get('via') or _domain(s.get('url','')) or s.get('title','') + if via and via not in names:names.append(via) + return 'Nguồn tham khảo: '+', '.join(names[:5]) if names else 'Nguồn tham khảo: tổng hợp internet' + + +def _source_badge(post): + sources=post.get('sources') or [] + for s in sources: + via=s.get('via') or _domain(s.get('url','')) + if via:return via + return _domain(post.get('url','')) or post.get('source') or 'VNEWS' + + +def _collect_all_images(data): + imgs=[] + def add(u): + u=(u or '').strip() + if not u or u.startswith('data:') or 'base64' in u:return + if u.startswith('//'):u='https:'+u + if u not in imgs:imgs.append(u) + add(data.get('image') or data.get('og_image') or data.get('img')) + for u in data.get('images') or []:add(u) + for b in data.get('body') or []: + if isinstance(b,dict) and b.get('type')=='img':add(b.get('src')) + return imgs[:20] + + +def _scrape_url_with_images(url): + data=base.scrape_any_url(url) + # extra pass: collect every useful image from original HTML, because some readers only return one image + try: + import requests + from bs4 import BeautifulSoup + r=requests.get(url,headers=base.HEADERS,timeout=18);r.encoding='utf-8' + soup=BeautifulSoup(r.text,'lxml') + extra=[] + for im in soup.find_all('img'): + src=im.get('data-src') or im.get('data-original') or im.get('data-lazy-src') or im.get('src') or '' + if src.startswith('//'):src='https:'+src + if src and 'base64' not in src and src not in extra: + # skip tiny icons/logos as much as possible + low=src.lower() + if any(x in low for x in ['logo','icon','avatar','sprite']): + continue + extra.append(src) + if len(extra)>=20:break + data['images']=_collect_all_images(data)+[u for u in extra if u not in _collect_all_images(data)] + except Exception: + data['images']=_collect_all_images(data) + data['images']=_collect_all_images(data) + if data['images'] and not data.get('image'): + data['image']=data['images'][0] + return data + + +def rich_context(topic, limit=5): + try: ctx,sources=base.web_context(topic, limit=limit) + except Exception: ctx,sources='',[] + rich=[];rs=[];seen=set() + for s in (sources or [])[:limit*2]: + url=s.get('url') or '' + if not url.startswith('http') or url in seen:continue + seen.add(url) + try: + data=base.scrape_any_url(url) + raw=(data.get('summary','')+'\n'+data.get('text','')).strip() + if len(raw)<180:continue + title=data.get('title') or s.get('title') or url + via=data.get('via') or s.get('via') or _domain(url) + rich.append(f"### {title} ({via})\n{raw[:2600]}") + rs.append({'title':title,'url':url,'excerpt':raw[:700],'via':via}) + if len(rich)>=limit:break + except Exception:continue + if rich:return '\n\n'.join(rich),rs + return ctx or f'Chủ đề: {topic}', sources or [] + + +def postprocess(text): + if hasattr(old,'_postprocess_ai_text'): + out=old._postprocess_ai_text(text, max_units=7) + else: + out=clean(text) + # keep wall text readable, but ensure short generation later won't show bullets + return out + + +# Remove old routes we must override. +_PATCH={('/api/topic_post','POST'),('/api/url_wall','POST'),('/api/rewrite_share','POST'),('/api/ai/url','POST'),('/api/ai/short/{post_id}','POST'),('/api/ai/short-file/{file_id}','GET'),('/','GET')} +app.router.routes=[r for r in app.router.routes if not any(getattr(r,'path',None)==p and m in getattr(r,'methods',set()) for p,m in _PATCH)] + + +@app.post('/api/url_wall') +async def url_wall_only(request:Request): + body=await request.json();url=base._clean_text(body.get('url','')) + if not url.startswith('http'):return JSONResponse({'error':'missing url'},status_code=400) + try:data=_scrape_url_with_images(url) + except Exception as e:return JSONResponse({'error':'Không scrape được URL: '+str(e)[:180]},status_code=422) + raw=(data.get('summary','')+'\n'+data.get('text','')).strip() + if len(raw)<120:return JSONResponse({'error':'URL không có đủ nội dung để tóm tắt'},status_code=422) + prompt=f"""Tóm tắt bài viết nguồn dưới đây để đăng lên Tường AI VNEWS. + +Yêu cầu bắt buộc: +- Chỉ tóm tắt nội dung chính, không viết lại toàn bộ bài. +- Ngắn gọn, cụ thể, dễ hiểu. +- Không lặp lại ý và không thêm chi tiết ngoài nguồn. +- Tối đa 5 ý chính hoặc 2 đoạn ngắn. +- Tránh dùng dấu đầu dòng nếu không thật cần thiết. + +Tiêu đề gốc: {data.get('title','')} +Nguồn: {data.get('via','') or _domain(url)} +Nội dung gốc: +{raw[:16000]}""" + text=await base.qwen_generate(prompt,image_url=(data.get('image') or None),max_tokens=900) + if not text:text=old._fallback_summary_from_prompt(prompt,max_units=5) if hasattr(old,'_fallback_summary_from_prompt') else raw[:900] + text=postprocess(text) + src=[{'title':data.get('title'), 'url':url, 'excerpt':raw[:500], 'via':data.get('via') or _domain(url)}] + if 'Nguồn tham khảo:' not in text:text+='\n\n'+source_line(src) + images=_collect_all_images(data) + post=base.make_post(data.get('title') or 'Bài viết',text,images[0] if images else (data.get('image') or ''),url,'url',sources=src) + post['images']=images + posts=base._load_ai_wall();posts.insert(0,post);base._save_ai_wall(posts) + return JSONResponse({'post':post}) + + +@app.post('/api/rewrite_share') +async def rewrite_share_url_only(request:Request): + return await url_wall_only(request) + + +@app.post('/api/ai/url') +async def ai_url_compat(request:Request): + return await url_wall_only(request) + + +@app.post('/api/topic_post') +async def topic_disabled(request:Request): + return JSONResponse({'error':'Đã tắt tạo bài theo chủ đề. Vui lòng dán URL bài viết để AI tóm tắt.'},status_code=410) + + +def split_segments(post,max_segments=8): + text=clean(post.get('text') or post.get('title') or '') + text=re.sub(r'Nguồn tham khảo:.*$','',text,flags=re.I|re.S).strip() + lines=[] + for ln in text.splitlines(): + ln=_strip_bullet_prefix(ln) + if len(ln)>=18:lines.append(ln) + if len(lines)<2: + lines=[_strip_bullet_prefix(s) for s in re.split(r'(?<=[\.\!\?])\s+',text) if len(_strip_bullet_prefix(s))>=25] + segs=[];cur='' + for ln in lines: + ln=_strip_bullet_prefix(ln) + if not ln:continue + if len(cur)+len(ln)<180:cur=(cur+' '+ln).strip() + else: + if cur:segs.append(_strip_bullet_prefix(cur)) + cur=ln + if cur:segs.append(_strip_bullet_prefix(cur)) + return segs[:max_segments] or [_strip_bullet_prefix(post.get('title','VNEWS'))] + + +def wrap_text(draw,text,font,maxw,max_lines): + words=clean(text).split();lines=[];cur='' + for w in words: + test=(cur+' '+w).strip() + try:width=draw.textbbox((0,0),test,font=font)[2] + except Exception:width=len(test)*20 + if width<=maxw:cur=test + else: + if cur:lines.append(cur) + cur=w + if len(lines)>=max_lines:break + if cur and len(lines)tr:nh=target[1];nw=int(nh*ratio) + else:nw=target[0];nh=int(nw/ratio) + im=im.resize((nw,nh));left=(nw-target[0])//2;top=(nh-target[1])//2 + bg.paste(im.crop((left,top,left+target[0],top+target[1])),(0,0)) + except Exception:pass + draw=ImageDraw.Draw(bg) + try: + fb=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',58) + ft=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',38) + fs=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf',30) + fsmall=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',28) + except Exception:fb=ft=fs=fsmall=None + # source badge on top image corner + badge='Nguồn: '+_source_badge(post) + try: + b=draw.textbbox((0,0),badge,font=fsmall);bw=b[2]-b[0];bh=b[3]-b[1] + except Exception: + bw=len(badge)*16;bh=34 + bx=W-bw-42;by=24 + draw.rounded_rectangle((bx-16,by-8,W-24,by+bh+14),radius=18,fill=(0,0,0,170)) + draw.text((bx,by),badge,fill=(255,255,255),font=fsmall) + # bottom text area + draw.rectangle((0,hero_h-20,W,H),fill=(12,12,12)) + # progress bars centered + total_w=total*38-14;start=(W-total_w)//2 + for i in range(total): + fill=(92,184,122) if i==idx else (70,70,70) + draw.rounded_rectangle((start+i*38,820,start+i*38+24,832),radius=6,fill=fill) + brand='VNEWS AI SHORT' + try: + bb=draw.textbbox((0,0),brand,font=ft);tx=(W-(bb[2]-bb[0]))//2 + except Exception:tx=360 + draw.text((tx,870),brand,fill=(110,231,143),font=ft) + clean_seg=_strip_bullet_prefix(seg) + lines=wrap_text(draw,clean_seg,fb,W-120,8) + block_h=len(lines)*74 + y=max(980, 1250-block_h//2) + _draw_center(draw,lines,fb,y,(255,255,255),W,74) + # small title centered near bottom + title_lines=wrap_text(draw,_strip_bullet_prefix(post.get('title','')),fs,W-120,3) + y2=1640 + draw.line((80,y2-26,W-80,y2-26),fill=(70,70,70),width=2) + _draw_center(draw,title_lines,fs,y2,(220,220,220),W,42) + bg.save(out_path,quality=92) + + +def make_tts(text,voice,out_path): + v={'nam':'vi-VN-NamMinhNeural','male':'vi-VN-NamMinhNeural','nu':'vi-VN-HoaiMyNeural','female':'vi-VN-HoaiMyNeural','mien-nam':'vi-VN-HoaiMyNeural'}.get(voice,'vi-VN-HoaiMyNeural') + text=_strip_bullet_prefix(text) + try:subprocess.run(['python','-m','edge_tts','--voice',v,'--text',text,'--write-media',out_path],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=160) + except Exception: + tld='com.vn' if voice in ('nu','female','mien-nam') else 'com' + try:base.gTTS(text,lang='vi',tld=tld,slow=False).save(out_path) + except TypeError:base.gTTS(text,lang='vi',slow=False).save(out_path) + + +@app.post('/api/ai/short/{post_id}') +async def short_segments(post_id:str,request:Request): + try:body=await request.json() + except Exception:body={} + voice=str(body.get('voice','nu')).lower().strip();emotion=str(body.get('emotion','neutral')).lower().strip();speed=max(0.85,min(1.35,float(body.get('speed',1.2) or 1.2))) + posts=base._load_ai_wall();post=next((p for p in posts if str(p.get('id'))==str(post_id)),None) + if not post:return JSONResponse({'error':'post not found'},status_code=404) + segs=split_segments(post,8) + os.makedirs(base.SHORTS_DIR,exist_ok=True);suffix=f'_{voice}_{emotion}_{str(speed).replace(".","p")}_centered_source_nobullet' + out=os.path.join(base.SHORTS_DIR,base._safe_name(post_id+suffix)+'.mp4') + if os.path.exists(out):post['video']='/api/ai/short-file/'+post_id+suffix;base._save_ai_wall(posts);return JSONResponse({'video':post['video'],'segments':len(segs),'subtitles':False}) + work=os.path.join(base.SHORTS_DIR,base._safe_name(post_id+suffix));os.makedirs(work,exist_ok=True) + img=os.path.join(work,'image.jpg');base._download_image(post.get('img'),post.get('title','AI news'),img) + clips=[] + try: + for i,seg in enumerate(segs): + frame=os.path.join(work,f'f{i}.jpg');aud=os.path.join(work,f'a{i}.mp3');aud2=os.path.join(work,f'a{i}_fast.mp3');clip=os.path.join(work,f'c{i}.mp4') + seg=_strip_bullet_prefix(seg) + make_frame(post,seg,i,len(segs),img,frame) + prefix={'urgent':'Tin nhanh.','warm':'Câu chuyện đáng chú ý.','serious':'Bản tin nghiêm túc.','energetic':'Cập nhật nổi bật.'}.get(emotion,'') + spoken=(prefix+' '+seg).strip() if i==0 and prefix else seg + make_tts(spoken,voice,aud) + subprocess.run(['ffmpeg','-y','-i',aud,'-filter:a',f'atempo={speed}','-vn',aud2],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=120) + subprocess.run(['ffmpeg','-y','-loop','1','-i',frame,'-i',aud2,'-shortest','-c:v','libx264','-tune','stillimage','-pix_fmt','yuv420p','-c:a','aac','-b:a','128k','-vf','scale=1080:1920',clip],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=180) + clips.append(clip) + lf=os.path.join(work,'list.txt') + with open(lf,'w',encoding='utf-8') as f: + for c in clips:f.write("file '{}".format(c.replace("'","'\\''"))+"'\n") + subprocess.run(['ffmpeg','-y','-f','concat','-safe','0','-i',lf,'-c','copy',out],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=240) + post['video']='/api/ai/short-file/'+post_id+suffix;post['short_subtitles']=False;post['short_segments']=segs;post['short_speed']=speed;base._save_ai_wall(posts) + return JSONResponse({'video':post['video'],'segments':len(segs),'subtitles':False}) + except Exception as e:return JSONResponse({'error':'Không tạo được shorts: '+str(e)[:200]},status_code=500) + + +@app.get('/api/ai/short-file/{file_id}') +def short_file(file_id:str): + path=os.path.join(base.SHORTS_DIR,base._safe_name(file_id)+'.mp4') + if not os.path.exists(path):return JSONResponse({'error':'not found'},status_code=404) + return FileResponse(path,media_type='video/mp4',filename=f'vnews-ai-{file_id}.mp4') + + +# Rebuild / with old UI injection plus final UI overrides. +app.router.routes=[r for r in app.router.routes if not (getattr(r,'path',None)=='/' and 'GET' in getattr(r,'methods',set()))] +@app.get('/') +async def index_runtime(): + with open('/app/static/index.html','r',encoding='utf-8') as f:html=f.read() + inject=getattr(old,'PATCH_INJECT','')+r''' + + +''' + return HTMLResponse(html.replace('',inject+'\n') if '' in html else html+inject) diff --git a/ai_runtime_final.py b/ai_runtime_final.py new file mode 100644 index 0000000000000000000000000000000000000000..8dce363f4414b76f00d41b1e4cccc2c2c3809d78 --- /dev/null +++ b/ai_runtime_final.py @@ -0,0 +1,315 @@ +"""Final runtime overrides for VNEWS AI UI, article-only images, shareable AI wall, and robust Vietnamese shorts.""" +import os, re, requests, subprocess, time +from urllib.parse import urlparse, quote +import ai_runtime as rt +from ai_runtime import app +import ai_ext as base +from fastapi import Request, Query +from fastapi.responses import HTMLResponse, JSONResponse, FileResponse +try: + from PIL import Image, ImageDraw, ImageFont +except Exception: + Image = ImageDraw = ImageFont = None + +RESTORE_INDEX_URL = "https://huggingface.co/spaces/bep40/vnews/raw/restore-33c3dda/static/index.html" +SPACE_URL = "https://bep40-vnews.hf.space" +DEFAULT_IMG = "https://s1.vnecdn.net/vnexpress/restruct/i/v9505/logo_default.jpg" + +# Only voices that support Vietnamese reliably. Extra labels map to these Vietnamese neural voices. +VN_VOICES = { + "nu": "vi-VN-HoaiMyNeural", "female": "vi-VN-HoaiMyNeural", "hoaimy": "vi-VN-HoaiMyNeural", + "nu-tre": "vi-VN-HoaiMyNeural", "nu-truyen-cam": "vi-VN-HoaiMyNeural", "nu-tin-nhanh": "vi-VN-HoaiMyNeural", + "nam": "vi-VN-NamMinhNeural", "male": "vi-VN-NamMinhNeural", "namminh": "vi-VN-NamMinhNeural", + "nam-tram": "vi-VN-NamMinhNeural", "nam-ban-tin": "vi-VN-NamMinhNeural", "nam-nang-dong": "vi-VN-NamMinhNeural", +} + + +def clean(s): + import html as html_lib + return re.sub(r"\s+", " ", html_lib.unescape(s or "")).strip() + + +def _domain(url): + try:return urlparse(url or '').netloc.replace('www.','') + except Exception:return '' + + +def _strip_bullet_prefix(s): + return clean(re.sub(r'^[\s•\-\*·▪▫●○\d\.\)\(]+', '', s or '')) + + +def _source_badge_url_first(post): + d=_domain(post.get('url','')) + if d:return d + for s in post.get('sources') or []: + d=_domain(s.get('url','')) + if d:return d + return 'VNEWS' + + +def _abs_url(src, base_url): + if not src:return '' + src=src.strip() + if src.startswith('//'):return 'https:'+src + if src.startswith('/'): + try: + p=urlparse(base_url);return f'{p.scheme}://{p.netloc}{src}' + except Exception:return src + return src + + +def _article_content_block(soup): + for tag in soup.find_all(['script','style','nav','footer','aside','form','noscript','iframe']):tag.decompose() + # Aggressively remove related/ad/recommend containers before image collection. + bad_re=re.compile(r'(related|relate|recommend|suggest|sidebar|ads|advert|popular|more|xem-them|xemthem|tin-lien-quan|tinlienquan|doc-them|docthem|other-news|news-other|article-related|box-tin|box_related|story-related|recommend-news|same-category|cate-list|news-list|most-view|banner|qc|quang-cao|sponsor)',re.I) + for el in list(soup.find_all(True)): + cls=' '.join(el.get('class',[])); eid=el.get('id',''); role=el.get('role','') + if bad_re.search(cls) or bad_re.search(eid) or bad_re.search(role): + el.decompose() + selectors=['article','main article','.article-content','.article__body','.article-body','.article-detail','.detail-content','.content-detail','.singular-content','.news-content','.post-content','.entry-content','.knc-content','.fck_detail','.cms-body','.story-body','[class*=article-content]','[class*=detail-content]','[class*=singular-content]'] + for sel in selectors: + el=soup.select_one(sel) + if el and (len(el.find_all('p'))>=2 or len(el.find_all(['figure','picture','img']))>=1):return el + best=None;score=0 + for el in soup.find_all(['article','main','section','div']): + ps=el.find_all('p');imgs=el.find_all('img');txt=' '.join(p.get_text(' ',strip=True) for p in ps) + sc=len(ps)*120+len(imgs)*10+min(len(txt),4500) + cls=' '.join(el.get('class',[])).lower() + if any(k in cls for k in ['article','content','detail','post','entry','story']):sc+=800 + if sc>score:best=el;score=sc + return best or soup + + +def _image_is_likely_article(im, src): + low=(src or '').lower() + if not src or src.startswith('data:') or 'base64' in low:return False + if any(x in low for x in ['logo','icon','avatar','sprite','banner','ads','advert','tracking','pixel','social','share','author','thumb-related']):return False + alt=(im.get('alt') or im.get('title') or '').lower() + if any(x in alt for x in ['logo','avatar','quảng cáo','advertisement','banner']):return False + try: + w=int(re.sub(r'\D','',str(im.get('width') or '0')) or 0);h=int(re.sub(r'\D','',str(im.get('height') or '0')) or 0) + if (w and w<220) or (h and h<140):return False + except Exception:pass + return True + + +def _article_only_images(url): + """Collect images only inside main article content. If uncertain, return fewer/no images rather than related/ad images.""" + imgs=[] + try: + from bs4 import BeautifulSoup + r=requests.get(url,headers=getattr(base,'HEADERS',{}),timeout=18);r.encoding='utf-8' + soup=BeautifulSoup(r.text,'lxml') + block=_article_content_block(soup) + candidates=[] + # Prefer figure/picture under article body; then direct img in body. + for el in block.find_all(['figure','picture'],recursive=True): + im=el.find('img') + if im:candidates.append(im) + for im in block.find_all('img',recursive=True): + if im not in candidates:candidates.append(im) + seen=set() + for im in candidates: + src=(im.get('data-src') or im.get('data-original') or im.get('data-lazy-src') or im.get('data-srcset') or im.get('srcset') or im.get('src') or '') + if ',' in src:src=src.split(',')[0].strip().split(' ')[0] + else:src=src.strip().split(' ')[0] + src=_abs_url(src,url) + if src in seen or not _image_is_likely_article(im,src):continue + # parent text guard: skip images from any remaining related block + parent_txt=' '.join((im.parent.get('class',[]) if im.parent else []))+' '+(im.parent.get('id','') if im.parent else '') + if re.search(r'(related|recommend|tin-lien-quan|doc-them|xem-them|popular|ads|banner)',parent_txt,re.I):continue + seen.add(src);imgs.append(src) + if len(imgs)>=20:break + # Use og:image ONLY as article main image fallback when no body image found. + if not imgs: + og=soup.find('meta',property='og:image') or soup.find('meta',attrs={'name':'twitter:image'}) + if og: + src=_abs_url(og.get('content',''),url) + if src and 'logo' not in src.lower() and 'banner' not in src.lower():imgs.append(src) + except Exception:pass + return imgs[:20] + + +def _scrape_url_article_only(url): + data=base.scrape_any_url(url) + imgs=_article_only_images(url) + data['images']=imgs + if imgs:data['image']=imgs[0] + else:data['image']='' + return data + + +def _blank_image(path, title='VNEWS'): + if Image is None:return None + im=Image.new('RGB',(1080,760),(24,48,36));draw=ImageDraw.Draw(im) + try:f=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',48) + except Exception:f=None + draw.text((60,330),clean(title)[:40] or 'VNEWS',fill=(255,255,255),font=f) + im.save(path,quality=90);return path + + +def _download_image_safe(url, fallback_title, out_path): + if url: + try: + r=requests.get(url,headers=getattr(base,'HEADERS',{}),timeout=18) + if r.status_code==200 and len(r.content)>1200: + with open(out_path,'wb') as f:f.write(r.content) + # verify PIL opens it + if Image: + Image.open(out_path).verify() + return out_path + except Exception:pass + try: + return base._download_image('',fallback_title,out_path) + except Exception: + return _blank_image(out_path,fallback_title) + + +def final_make_tts(text,voice,out_path): + text=_strip_bullet_prefix(text) or 'Bản tin VNEWS.' + # Only Vietnamese voices. Unknown choices fall back to Vietnamese female. + edge_voice=VN_VOICES.get(str(voice or '').lower().strip(), 'vi-VN-HoaiMyNeural') + for ev in [edge_voice, 'vi-VN-HoaiMyNeural', 'vi-VN-NamMinhNeural']: + try: + subprocess.run(['python','-m','edge_tts','--voice',ev,'--text',text,'--write-media',out_path],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=180) + if os.path.exists(out_path) and os.path.getsize(out_path)>1000:return out_path + except Exception:pass + try: + base.gTTS(text,lang='vi',tld='com.vn',slow=False).save(out_path) + if os.path.exists(out_path) and os.path.getsize(out_path)>1000:return out_path + except Exception:pass + # Last-resort silent audio guarantees short generation succeeds. + subprocess.run(['ffmpeg','-y','-f','lavfi','-i','anullsrc=channel_layout=stereo:sample_rate=44100','-t','3','-q:a','9','-acodec','libmp3lame',out_path],stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=30) + return out_path + + +def _draw_center(draw, lines, font, y, fill, W, line_h): + for ln in lines: + try:box=draw.textbbox((0,0),ln,font=font);tw=box[2]-box[0] + except Exception:tw=len(ln)*24 + draw.text((max(30,(W-tw)//2),y),ln,fill=fill,font=font);y+=line_h + return y + + +def final_make_frame(post,seg,idx,total,img_path,out_path): + if Image is None:return rt.make_frame(post,seg,idx,total,img_path,out_path) + W,H=1080,1920;hero_h=760;bg=Image.new('RGB',(W,H),(12,12,12)) + try: + im=Image.open(img_path).convert('RGB');ratio=im.width/max(1,im.height);tr=W/hero_h + if ratio>tr:nh=hero_h;nw=int(nh*ratio) + else:nw=W;nh=int(nw/ratio) + im=im.resize((nw,nh));left=(nw-W)//2;top=(nh-hero_h)//2;bg.paste(im.crop((left,top,left+W,top+hero_h)),(0,0)) + except Exception:pass + draw=ImageDraw.Draw(bg) + try: + fb=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',58);ft=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',38);fs=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf',30);fsmall=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',28) + except Exception:fb=ft=fs=fsmall=None + badge='Nguồn: '+_source_badge_url_first(post) + try:b=draw.textbbox((0,0),badge,font=fsmall);bw=b[2]-b[0];bh=b[3]-b[1] + except Exception:bw=len(badge)*16;bh=34 + bx=W-bw-42;by=24;draw.rounded_rectangle((bx-16,by-8,W-24,by+bh+14),radius=18,fill=(0,0,0));draw.text((bx,by),badge,fill=(255,255,255),font=fsmall) + draw.rectangle((0,hero_h-20,W,H),fill=(12,12,12)) + total=max(1,total);total_w=total*38-14;start=(W-total_w)//2 + for i in range(total):draw.rounded_rectangle((start+i*38,820,start+i*38+24,832),radius=6,fill=(92,184,122) if i==idx else (70,70,70)) + brand='VNEWS AI SHORT' + try:bb=draw.textbbox((0,0),brand,font=ft);tx=(W-(bb[2]-bb[0]))//2 + except Exception:tx=360 + draw.text((tx,870),brand,fill=(110,231,143),font=ft) + seg=_strip_bullet_prefix(seg);lines=rt.wrap_text(draw,seg,fb,W-120,8);y=max(980,1250-(len(lines)*74)//2);_draw_center(draw,lines,fb,y,(255,255,255),W,74) + title_lines=rt.wrap_text(draw,_strip_bullet_prefix(post.get('title','')),fs,W-120,3);y2=1640;draw.line((80,y2-26,W-80,y2-26),fill=(70,70,70),width=2);_draw_center(draw,title_lines,fs,y2,(220,220,220),W,42) + bg.save(out_path,quality=92) + +# Monkey patches for old functions. +rt.make_frame=final_make_frame;rt.make_tts=final_make_tts;rt._source_badge=_source_badge_url_first + +# Override endpoints. +_PATCH={('/api/url_wall','POST'),('/api/rewrite_share','POST'),('/api/ai/url','POST'),('/api/ai/short/{post_id}','POST'),('/','GET'),('/aw','GET')} +app.router.routes=[r for r in app.router.routes if not any(getattr(r,'path',None)==p and m in getattr(r,'methods',set()) for p,m in _PATCH)] + +@app.post('/api/url_wall') +async def final_url_wall(request:Request): + body=await request.json();url=base._clean_text(body.get('url','')) + if not url.startswith('http'):return JSONResponse({'error':'missing url'},status_code=400) + try:data=_scrape_url_article_only(url) + except Exception as e:return JSONResponse({'error':'Không scrape được URL: '+str(e)[:180]},status_code=422) + raw=(data.get('summary','')+'\n'+data.get('text','')).strip() + if len(raw)<120:return JSONResponse({'error':'URL không có đủ nội dung để tóm tắt'},status_code=422) + prompt=f"""Tóm tắt bài viết nguồn dưới đây để đăng lên Tường AI VNEWS. + +Yêu cầu: +- Chỉ tóm tắt nội dung chính, không viết lại toàn bộ bài. +- Ngắn gọn, cụ thể, dễ hiểu. +- Không lặp ý, không thêm chi tiết ngoài nguồn. +- Tối đa 5 ý chính hoặc 2 đoạn ngắn. +- Hạn chế dùng dấu đầu dòng. + +Tiêu đề gốc: {data.get('title','')} +Nguồn: {_domain(url)} +Nội dung gốc: +{raw[:16000]}""" + text=await base.qwen_generate(prompt,image_url=(data.get('image') or None),max_tokens=900) + if not text:text=rt.old._fallback_summary_from_prompt(prompt,max_units=5) if hasattr(rt.old,'_fallback_summary_from_prompt') else raw[:900] + text=rt.postprocess(text) if hasattr(rt,'postprocess') else text + src=[{'title':data.get('title'), 'url':url, 'excerpt':raw[:500], 'via':_domain(url)}] + if 'Nguồn tham khảo:' not in text:text+='\n\n'+rt.source_line(src) + imgs=data.get('images') or [] + post=base.make_post(data.get('title') or 'Bài viết',text,imgs[0] if imgs else '',url,'url',sources=src) + post['images']=imgs + posts=base._load_ai_wall();posts.insert(0,post);base._save_ai_wall(posts) + return JSONResponse({'post':post}) + +@app.post('/api/rewrite_share') +async def final_rewrite_share(request:Request):return await final_url_wall(request) +@app.post('/api/ai/url') +async def final_ai_url(request:Request):return await final_url_wall(request) + +@app.post('/api/ai/short/{post_id}') +async def final_short(post_id:str,request:Request): + try:body=await request.json() + except Exception:body={} + voice=str(body.get('voice','nu')).lower().strip();emotion=str(body.get('emotion','neutral')).lower().strip();speed=max(0.85,min(1.35,float(body.get('speed',1.2) or 1.2))) + posts=base._load_ai_wall();post=next((p for p in posts if str(p.get('id'))==str(post_id)),None) + if not post:return JSONResponse({'error':'post not found'},status_code=404) + segs=rt.split_segments(post,8) if hasattr(rt,'split_segments') else [_strip_bullet_prefix(post.get('text') or post.get('title') or 'VNEWS')] + imgs=[u for u in (post.get('images') or []) if u] or ([post.get('img')] if post.get('img') else []) + os.makedirs(base.SHORTS_DIR,exist_ok=True);suffix=f'_{voice}_{emotion}_{str(speed).replace(".","p")}_articleimgs_vivoice' + out=os.path.join(base.SHORTS_DIR,base._safe_name(post_id+suffix)+'.mp4') + if os.path.exists(out): + post['video']='/api/ai/short-file/'+post_id+suffix;base._save_ai_wall(posts);return JSONResponse({'video':post['video'],'segments':len(segs),'subtitles':False}) + work=os.path.join(base.SHORTS_DIR,base._safe_name(post_id+suffix));os.makedirs(work,exist_ok=True) + clips=[] + try: + for i,seg in enumerate(segs): + img_url=imgs[i % len(imgs)] if imgs else '' + img=os.path.join(work,f'image_{i}.jpg');frame=os.path.join(work,f'f{i}.jpg');aud=os.path.join(work,f'a{i}.mp3');aud2=os.path.join(work,f'a{i}_fast.mp3');clip=os.path.join(work,f'c{i}.mp4') + _download_image_safe(img_url,post.get('title','AI news'),img) + seg=_strip_bullet_prefix(seg);final_make_frame(post,seg,i,len(segs),img,frame) + prefix={'urgent':'Tin nhanh.','warm':'Câu chuyện đáng chú ý.','serious':'Bản tin nghiêm túc.','energetic':'Cập nhật nổi bật.'}.get(emotion,'') + spoken=(prefix+' '+seg).strip() if i==0 and prefix else seg + final_make_tts(spoken,voice,aud) + try:subprocess.run(['ffmpeg','-y','-i',aud,'-filter:a',f'atempo={speed}','-vn',aud2],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=120) + except Exception:aud2=aud + try: + subprocess.run(['ffmpeg','-y','-loop','1','-i',frame,'-i',aud2,'-shortest','-c:v','libx264','-tune','stillimage','-pix_fmt','yuv420p','-c:a','aac','-b:a','128k','-vf','scale=1080:1920',clip],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=180) + except Exception: + # last-resort visual-only 4s clip + subprocess.run(['ffmpeg','-y','-loop','1','-t','4','-i',frame,'-f','lavfi','-i','anullsrc=channel_layout=stereo:sample_rate=44100','-shortest','-c:v','libx264','-pix_fmt','yuv420p','-c:a','aac','-vf','scale=1080:1920',clip],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=120) + clips.append(clip) + lf=os.path.join(work,'list.txt') + with open(lf,'w',encoding='utf-8') as f: + for c in clips:f.write("file '"+c.replace("","'\\''"))+"'\n") + subprocess.run(['ffmpeg','-y','-f','concat','-safe','0','-i',lf,'-c','copy',out],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=240) + post['video']='/api/ai/short-file/'+post_id+suffix;post['short_subtitles']=False;post['short_segments']=segs;post['short_speed']=speed;base._save_ai_wall(posts) + return JSONResponse({'video':post['video'],'segments':len(segs),'subtitles':False}) + except Exception as e:return JSONResponse({'error':'Không tạo được shorts: '+str(e)[:220]},status_code=500) + +@app.get('/aw') +def ai_wall_share(post:str=Query(default=''), short:int=Query(default=0)): + posts=base._load_ai_wall();p=next((x for x in posts if str(x.get('id'))==str(post)),None) + if not p:return HTMLResponse(f'') + title=p.get('title') or 'VNEWS AI';img=p.get('img') or DEFAULT_IMG + desc=(p.get('text') or '')[:220] + return HTMLResponse(f'{title}') + +FINAL_INJECT = r''' + +
+ +''' + +@app.get('/') +async def index_final3(): + html=f2.f1._load_index_html();body=getattr(rt.old,'PATCH_INJECT','') + f2.f1.FINAL_INJECT + FINAL3_INJECT + return HTMLResponse(html.replace('',body+'\n') if '' in html else html+body) diff --git a/ai_runtime_final4.py b/ai_runtime_final4.py new file mode 100644 index 0000000000000000000000000000000000000000..e3ae1b8a86a071ca71e71bf8adcc76ed2ede8369 --- /dev/null +++ b/ai_runtime_final4.py @@ -0,0 +1,185 @@ +"""Final4 runtime: fix topic button visibility, shorts home feed, AI asking for videos/articles.""" +import re, time, json, os, requests +from urllib.parse import urlparse +import ai_runtime_final3 as f3 +from ai_runtime_final3 import app, base, rt, HTMLResponse, JSONResponse, Request, Query +try: + import main as main_mod +except Exception: + main_mod=None + +AI_INTERACTIONS_FILE=f3.AI_INTERACTIONS_FILE +_SHORTS_CACHE={"t":0,"d":[]} +SHORT_CHANNELS=f3.SHORT_CHANNELS + + +def clean(s): + import html as html_lib + return re.sub(r"\s+"," ",html_lib.unescape(s or "")).strip() + + +def _domain(u): + try:return urlparse(u or '').netloc.replace('www.','') + except Exception:return '' + + +def _load_json(path,default): + try: + if os.path.exists(path): + with open(path,'r',encoding='utf-8') as f:return json.load(f) + except Exception:pass + return default + + +def _save_json(path,data): + try: + os.makedirs(os.path.dirname(path),exist_ok=True);tmp=path+'.tmp' + with open(tmp,'w',encoding='utf-8') as f:json.dump(data,f,ensure_ascii=False) + os.replace(tmp,path) + except Exception:pass + + +def _fallback_shorts(): + out=[];seen=set() + candidates=[] + try:candidates+=(getattr(main_mod,'SHORTS_FALLBACK',[]) or []) + except Exception:pass + try:candidates+=(getattr(rt,'SHORTS_FALLBACK',[]) or []) + except Exception:pass + # hard fallback if imports fail + hard=[('Lu_iCQ5YwNM','Công an lập hồ sơ xử lý người phụ nữ chửi bới, tát tài xế ô tô | Dân trí','baodantri7941'),('CwWvijF8BOA','Chú rể bật khóc nhận món quà bí mật người cha quá cố gửi 26 năm trước | Dân trí','baodantri7941'),('7Pd6vZ2Lz1M','Hành động ấm lòng trong tìm kiếm học sinh tử vong ở sông Lô | SKĐS','baosuckhoedoisongboyte'),('SlHLt_ZyPiE','Xử phạt người đàn ông xóa số điện thoại cứu hộ trên cao tốc Bắc - Nam | SKĐS','baosuckhoedoisongboyte')] + for vid,title,ch in hard: + candidates.append({'id':vid,'title':title,'channel':ch,'link':'https://www.youtube.com/watch?v='+vid,'img':'https://i.ytimg.com/vi/'+vid+'/hqdefault.jpg','source':'yt'}) + for v in candidates: + vid=v.get('id') or '' + if vid and vid not in seen: + seen.add(vid) + if not v.get('link'):v['link']='https://www.youtube.com/watch?v='+vid + if not v.get('img'):v['img']='https://i.ytimg.com/vi/'+vid+'/hqdefault.jpg' + v['source']='yt';out.append(v) + return out + + +def _fresh_shorts(): + items=[];seen=set() + for ch in SHORT_CHANNELS: + got=f3._youtube_shorts_ytdlp(ch,24) or f3._youtube_shorts_html(ch,24) + for v in got: + vid=v.get('id') + if vid and vid not in seen: + seen.add(vid);items.append(v) + for v in _fallback_shorts(): + vid=v.get('id') + if vid and vid not in seen: + seen.add(vid);items.append(v) + return items[:60] + +# Remove endpoints/root to override. +_PATCH={('/api/shorts','GET'),('/api/ai/interact','POST'),('/api/article/ask','POST'),('/','GET')} +app.router.routes=[r for r in app.router.routes if not any(getattr(r,'path',None)==p and m in getattr(r,'methods',set()) for p,m in _PATCH)] + +@app.get('/api/shorts') +def api_shorts_final4(refresh:int=Query(default=0)): + now=time.time() + if not refresh and _SHORTS_CACHE['d'] and now-_SHORTS_CACHE['t']<900:return JSONResponse(_SHORTS_CACHE['d']) + data=_fresh_shorts() + _SHORTS_CACHE.update({'t':now,'d':data}) + return JSONResponse(data) + +@app.post('/api/ai/interact') +async def ai_interact_final4(request:Request): + body=await request.json();pid=str(body.get('id','')).strip();kind=str(body.get('kind','wall')).strip();action=str(body.get('action','')).strip();text=clean(body.get('text',''));context=clean(body.get('context',''));title=clean(body.get('title','')) + if not pid:return JSONResponse({'error':'missing id'},status_code=400) + db=_load_json(AI_INTERACTIONS_FILE,{}) + key=kind+':'+pid + st=db.get(key) or {'views':0,'likes':0,'comments':[],'asks':[]} + if action=='view':st['views']=int(st.get('views',0))+1 + elif action=='like':st['likes']=int(st.get('likes',0))+1 + elif action=='comment' and text: + st.setdefault('comments',[]).insert(0,{'text':text[:240],'ts':int(time.time())});st['comments']=st['comments'][:80] + elif action=='ask' and text: + if kind in ('ai','short','wall'): + posts=base._load_ai_wall();p=next((x for x in posts if str(x.get('id'))==pid),{}) + title=title or p.get('title','');context=context or (p.get('text') or '') + # For YouTube shorts, frontend sends title/context because AI cannot watch video. + if not context:context=title or pid + prompt=f"""Bạn là trợ lý VNEWS. Trả lời chi tiết bằng tiếng Việt dựa trên thông tin có sẵn về video/bài viết. + +Tiêu đề/ngữ cảnh: {title} +Nội dung mô tả: {context[:5000]} + +Câu hỏi người dùng: {text} + +Yêu cầu: +- Nếu là video YouTube/Shorts và chỉ có tiêu đề, hãy nói rõ rằng bạn suy luận từ tiêu đề/mô tả, không khẳng định đã xem video. +- Trả lời cụ thể, có giải thích, không quá ngắn. +""" + ans=await base.qwen_generate(prompt,max_tokens=900) + if not ans:ans='AI chưa trả lời được lúc này. Bạn thử hỏi lại cụ thể hơn.' + st.setdefault('asks',[]).insert(0,{'q':text[:240],'a':ans[:1500],'ts':int(time.time())});st['asks']=st['asks'][:50] + db[key]=st;_save_json(AI_INTERACTIONS_FILE,db) + return JSONResponse({'stats':st}) + +@app.post('/api/article/ask') +async def article_ask(request:Request): + body=await request.json();url=clean(body.get('url',''));question=clean(body.get('question','')) + if not question:return JSONResponse({'error':'missing question'},status_code=400) + title='';raw='' + try: + data=None + if url and hasattr(f3.f2.f1,'_scrape_url_article_only'): + data=f3.f2.f1._scrape_url_article_only(url) + if not data and url:data=base.scrape_any_url(url) + if data: + title=data.get('title','');raw=(data.get('summary','')+'\n'+data.get('text','')).strip() + except Exception:pass + context=raw[:12000] if raw else clean(body.get('context',''))[:12000] + prompt=f"""Bạn là trợ lý đọc hiểu bài viết của VNEWS. Hãy trả lời chi tiết câu hỏi của người dùng dựa trên bài viết. + +Tiêu đề bài: {title} +Nội dung bài: +{context} + +Câu hỏi: {question} + +Yêu cầu: +- Trả lời bằng tiếng Việt. +- Dựa sát nội dung bài, nếu bài không có thông tin thì nói rõ. +- Giải thích chi tiết, có gạch đầu dòng khi hữu ích. +""" + ans=await base.qwen_generate(prompt,max_tokens=1200) + if not ans:ans='AI chưa trả lời được lúc này. Bạn thử hỏi lại hoặc rút gọn câu hỏi.' + return JSONResponse({'answer':ans,'title':title}) + +FINAL4_INJECT = r''' + + +''' + +@app.get('/') +async def index_final4(): + html=f3.f2.f1._load_index_html();body=getattr(rt.old,'PATCH_INJECT','')+f3.f2.f1.FINAL_INJECT+f3.FINAL3_INJECT+FINAL4_INJECT + return HTMLResponse(html.replace('',body+'\n') if '' in html else html+body) diff --git a/ai_runtime_final5.py b/ai_runtime_final5.py new file mode 100644 index 0000000000000000000000000000000000000000..cab91a90e41225b2d3a32db9c034dc05fafb4258 --- /dev/null +++ b/ai_runtime_final5.py @@ -0,0 +1,73 @@ +"""Final5 runtime: remove duplicate topic box, improve Qwen topic knowledge output, fix Shorts direct playback.""" +import re, time +from urllib.parse import quote +import ai_runtime_final4 as f4 +from ai_runtime_final4 import app, base, rt, HTMLResponse, JSONResponse, Request, Query + +# Remove topic/root endpoints to override. +_PATCH={('/api/topic_post','POST'),('/','GET')} +app.router.routes=[r for r in app.router.routes if not any(getattr(r,'path',None)==p and m in getattr(r,'methods',set()) for p,m in _PATCH)] + +def clean(s): + import html as html_lib + return re.sub(r"\s+"," ",html_lib.unescape(s or "")).strip() + +def _topic_image(topic): + try:return base.pollinations_image_url(topic) + except Exception:return "https://image.pollinations.ai/prompt/"+quote("Vietnamese educational editorial illustration "+topic)+"?width=1024&height=576&nologo=true" + +@app.post('/api/topic_post') +async def topic_post_knowledge(request:Request): + body=await request.json();topic=clean(body.get('topic','')) + if not topic:return JSONResponse({'error':'missing topic'},status_code=400) + img=_topic_image(topic) + prompt=f"""Người dùng muốn đăng một bài trên Tường AI về chủ đề: "{topic}". + +Hãy viết NGAY nội dung kiến thức/thông tin hữu ích về chủ đề đó, không lập dàn ý chung chung, không nói "có thể viết", không hướng dẫn cách viết. + +Yêu cầu đầu ra: +- Tiêu đề hấp dẫn, cụ thể. +- 1 đoạn mở đầu giải thích trực tiếp chủ đề là gì/vì sao đáng chú ý. +- 5-7 đoạn hoặc ý chính cung cấp kiến thức thực chất, ví dụ, bối cảnh, tác động, hiểu lầm thường gặp, điểm cần lưu ý. +- Nếu chủ đề là thể thao, hãy nói về bối cảnh, nhân vật/đội bóng, ý nghĩa chiến thuật hoặc lịch sử liên quan. +- Nếu chủ đề là công nghệ/khoa học/xã hội, hãy giải thích khái niệm, ứng dụng, rủi ro/lợi ích, ví dụ thực tế. +- Không bịa số liệu thời sự mới; nếu không chắc, dùng cách nói thận trọng. +- Viết như bài đăng hoàn chỉnh để đọc được ngay. +- Cuối bài thêm: Nguồn tham khảo: Qwen2.5-VL / kiến thức tổng hợp. +""" + text=await base.qwen_generate(prompt,image_url=img,max_tokens=1400) + if not text: + text=f"{topic}\n\n{topic} là một chủ đề có nhiều khía cạnh cần nhìn từ bối cảnh, ý nghĩa thực tế và tác động đối với người quan tâm. Bài viết này tóm lược các điểm quan trọng nhất để người đọc hiểu nhanh vấn đề, thay vì chỉ liệt kê tiêu đề hoặc dàn ý.\n\nNguồn tham khảo: Qwen2.5-VL / kiến thức tổng hợp." + post=base.make_post(topic,text,img,'','topic_qwen',sources=[{'title':'Qwen2.5-VL / kiến thức tổng hợp','url':'','via':'Qwen2.5-VL'}]) + post['images']=[img] + posts=base._load_ai_wall();posts.insert(0,post);base._save_ai_wall(posts) + return JSONResponse({'post':post}) + +FINAL5_INJECT=r''' + + +''' + +@app.get('/') +async def index_final5(): + html=f4.f3.f2.f1._load_index_html();body=getattr(rt.old,'PATCH_INJECT','')+f4.f3.f2.f1.FINAL_INJECT+f4.f3.FINAL3_INJECT+f4.FINAL4_INJECT+FINAL5_INJECT + return HTMLResponse(html.replace('',body+'\n') if '' in html else html+body) diff --git a/ai_runtime_final6.py b/ai_runtime_final6.py new file mode 100644 index 0000000000000000000000000000000000000000..4efc9c899924f859f2e644ac3e18c98919615c20 --- /dev/null +++ b/ai_runtime_final6.py @@ -0,0 +1,849 @@ +"""Final6: robust topic synthesis, stable shorts, hot topic hashtags. + +This runtime intentionally overrides only the topic/shorts/root endpoints from the restored app. +""" +import re, time, json, os, threading, html as html_lib +from urllib.parse import quote, urlparse, parse_qs, unquote +import requests +from bs4 import BeautifulSoup +import ai_runtime_final5 as f5 +from ai_runtime_final5 import app, rt, HTMLResponse, JSONResponse, Request, Query + +_PATCH={('/api/topic_post','POST'),('/api/shorts','GET'),('/api/hot_topics','GET'),('/api/topic_sources','GET'),('/','GET')} +app.router.routes=[r for r in app.router.routes if not any(getattr(r,'path',None)==p and m in getattr(r,'methods',set()) for p,m in _PATCH)] + +_TOPIC_CACHE={} +_HOT_CACHE={"t":0,"d":[]} +_SHORTS_CACHE_FINAL6={"t":0,"d":[]} +_TRANSLATE_CACHE_PATH="/data/title_vi_cache.json" if os.path.isdir('/data') else "/app/data/title_vi_cache.json" +_translate_lock=threading.Lock() +YOUTUBE_HANDLES=["baodantri7941","baosuckhoedoisongboyte"] +UA={"User-Agent":"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36","Accept-Language":"vi,en;q=0.8"} +STOP_WORDS=set('và của các những một được trong với cho tại sau trước khi không người việt nam hôm nay mới nhất nóng tin tức cập nhật'.split()) +TRUSTED_SITES=['vnexpress.net','dantri.com.vn','vietnamnet.vn','tuoitre.vn','thanhnien.vn','laodong.vn','vov.vn','vtv.vn','genk.vn','cafef.vn','thethaovanhoa.vn'] + +def clean(s):return re.sub(r"\s+"," ",html_lib.unescape(str(s or ""))).strip() +def _domain(u): + try:return urlparse(u or '').netloc.replace('www.','') + except Exception:return '' + +def _load_title_cache(): + try: + if os.path.exists(_TRANSLATE_CACHE_PATH): + with open(_TRANSLATE_CACHE_PATH,'r',encoding='utf-8') as f:return json.load(f) + except Exception:pass + return {} +def _save_title_cache(db): + try: + os.makedirs(os.path.dirname(_TRANSLATE_CACHE_PATH),exist_ok=True);tmp=_TRANSLATE_CACHE_PATH+'.tmp' + with open(tmp,'w',encoding='utf-8') as f:json.dump(db,f,ensure_ascii=False) + os.replace(tmp,_TRANSLATE_CACHE_PATH) + except Exception:pass + +def _looks_vietnamese(s): + s=s or '' + if re.search(r'[àáạảãâầấậẩẫăằắặẳẵèéẹẻẽêềếệểễìíịỉĩòóọỏõôồốộổỗơờớợởỡùúụủũưừứựửữỳýỵỷỹđ]',s,re.I):return True + low=' '+s.lower()+' ' + return any(w in low for w in [' và ',' của ',' người ',' tại ',' trong ',' với ',' không ',' được ',' công an ',' bệnh viện ',' học sinh ',' tài xế ',' bóng đá ',' tin tức ',' sức khỏe ']) +def _translate_title_vi(title): + title=clean(title) + if not title or _looks_vietnamese(title):return title + with _translate_lock: + db=_load_title_cache() + if title in db:return db[title] + vi=title + try: + r=requests.get('https://translate.googleapis.com/translate_a/single',params={'client':'gtx','sl':'auto','tl':'vi','dt':'t','q':title},headers=UA,timeout=8) + if r.status_code==200: + data=r.json();vi=''.join(part[0] for part in data[0] if part and part[0]).strip() or title + except Exception:pass + vi=clean(vi) + with _translate_lock: + db=_load_title_cache();db[title]=vi;_save_title_cache(db) + return vi + +# ===== Hot topics / hashtags ===== +def _keywords_from_title(title): + title=clean(re.sub(r'\s+-\s+.*$','',title)) + words=[w for w in re.findall(r'[A-Za-zÀ-ỹ0-9]+',title) if len(w)>2 and w.lower() not in STOP_WORDS] + phrases=[] + for n in (4,3,2): + for i in range(0,max(0,len(words)-n+1)): + ph=' '.join(words[i:i+n]).strip() + if len(ph)>=8:phrases.append(ph) + if words:phrases.append(' '.join(words[:5])) + return phrases[:4] + +def _hot_topics(): + now=time.time() + if _HOT_CACHE['d'] and now-_HOT_CACHE['t']<900:return _HOT_CACHE['d'] + topics=[];seen=set() + feeds=[ + 'https://news.google.com/rss?hl=vi&gl=VN&ceid=VN:vi', + 'https://news.google.com/rss/headlines/section/topic/NATION?hl=vi&gl=VN&ceid=VN:vi', + 'https://news.google.com/rss/headlines/section/topic/BUSINESS?hl=vi&gl=VN&ceid=VN:vi', + 'https://news.google.com/rss/headlines/section/topic/SPORTS?hl=vi&gl=VN&ceid=VN:vi', + 'https://news.google.com/rss/headlines/section/topic/TECHNOLOGY?hl=vi&gl=VN&ceid=VN:vi' + ] + for feed in feeds: + try: + r=requests.get(feed,headers=UA,timeout=10);r.encoding='utf-8' + soup=BeautifulSoup(r.text,'xml') + for it in soup.find_all('item')[:15]: + title=clean(it.find('title').get_text(' ',strip=True) if it.find('title') else '') + for kw in _keywords_from_title(title): + key=kw.lower() + if key not in seen and len(kw)<=60: + seen.add(key);topics.append({'label':'#'+re.sub(r'\s+','',kw.title()),'topic':kw}) + if len(topics)>=24:break + if len(topics)>=24:break + except Exception:pass + if len(topics)>=24:break + for kw in ['AI trong giáo dục','World Cup 2026','kinh tế Việt Nam','biến đổi khí hậu','giá vàng','bóng đá Việt Nam','an ninh mạng','xe điện','sức khỏe tinh thần','thị trường chứng khoán']: + if kw.lower() not in seen:topics.append({'label':'#'+re.sub(r'\s+','',kw.title()),'topic':kw}) + _HOT_CACHE.update({'t':now,'d':topics[:24]}) + return _HOT_CACHE['d'] +@app.get('/api/hot_topics') +def api_hot_topics():return JSONResponse({'topics':_hot_topics()}) + +# ===== Topic web research ===== +def _unwrap_ddg_href(href): + if not href:return '' + if href.startswith('//duckduckgo.com/l/?') or 'duckduckgo.com/l/?' in href: + qs=parse_qs(urlparse('https:'+href if href.startswith('//') else href).query) + return unquote(qs.get('uddg',[''])[0]) + return href + +def _ddg_search(query, limit=10): + items=[];seen=set() + try: + url='https://html.duckduckgo.com/html/?q='+quote(query) + r=requests.get(url,headers=UA,timeout=14);r.encoding='utf-8' + soup=BeautifulSoup(r.text,'lxml') + for res in soup.select('.result'): + a=res.select_one('.result__title a') or res.find('a',href=True) + if not a:continue + link=_unwrap_ddg_href(a.get('href',''));title=clean(a.get_text(' ',strip=True));snippet=clean((res.select_one('.result__snippet') or res).get_text(' ',strip=True)) + if not link.startswith('http') or link in seen:continue + if any(bad in link for bad in ['duckduckgo.com','youtube.com','facebook.com','tiktok.com','twitter.com','x.com']):continue + seen.add(link);items.append({'title':title,'url':link,'source':_domain(link),'snippet':snippet}) + if len(items)>=limit:break + except Exception:pass + return items + +def _google_news_items(topic, limit=8): + items=[];seen=set() + try: + rss='https://news.google.com/rss/search?q='+quote(topic)+'&hl=vi&gl=VN&ceid=VN:vi' + r=requests.get(rss,headers=UA,timeout=12);r.encoding='utf-8' + soup=BeautifulSoup(r.text,'xml') + for it in soup.find_all('item')[:limit*2]: + title=clean(it.find('title').get_text(' ',strip=True) if it.find('title') else '') + link=clean(it.find('link').get_text(strip=True) if it.find('link') else '') + src=clean(it.find('source').get_text(' ',strip=True) if it.find('source') else _domain(link)) + if title and link and link not in seen: + seen.add(link);items.append({'title':title,'url':link,'source':src,'snippet':''}) + if len(items)>=limit:break + except Exception:pass + return items + +def _candidate_urls(topic): + seen=set();items=[] + queries=[topic+' tin tức Việt Nam', topic+' phân tích bối cảnh', topic+' site:vnexpress.net OR site:dantri.com.vn OR site:vietnamnet.vn'] + for q in queries: + for it in _ddg_search(q,8): + if it['url'] not in seen: + seen.add(it['url']);items.append(it) + if len(items)>=12:break + for site in TRUSTED_SITES[:8]: + for it in _ddg_search(f'{topic} site:{site}',3): + if it['url'] not in seen: + seen.add(it['url']);items.append(it) + for it in _google_news_items(topic,8): + if it['url'] not in seen: + seen.add(it['url']);items.append(it) + return items[:24] + +def _extract_article_text_bs(url, max_chars=9000): + try: + r=requests.get(url,headers=UA,timeout=16,allow_redirects=True) + if r.status_code>=400:return '' + r.encoding='utf-8';soup=BeautifulSoup(r.text,'lxml') + for tag in soup.find_all(['script','style','nav','footer','aside','form','noscript','iframe','svg']):tag.decompose() + candidates=[] + for sel in ['article','main','.article-content','.detail-content','.singular-content','.fck_detail','.content-detail','.entry-content','.story-body','.knc-content']: + el=soup.select_one(sel) + if el:candidates.append(el) + if not candidates:candidates=[soup.body or soup] + best=max(candidates,key=lambda el:len(el.find_all('p')) if el else 0) + ps=[] + for el in best.find_all(['p','h2','h3'],recursive=True): + t=clean(el.get_text(' ',strip=True)) + if len(t)>45 and not any(x in t.lower() for x in ['đăng ký nhận tin','theo dõi chúng tôi','chuyên mục','xem thêm','tin liên quan','advertisement']):ps.append(t) + if sum(len(x) for x in ps)>max_chars:break + return '\n'.join(ps)[:max_chars] + except Exception:return '' + +def _jina_read_text(url, max_chars=9000): + try: + ju='https://r.jina.ai/http://'+url + r=requests.get(ju,headers=UA,timeout=28);r.encoding='utf-8' + if r.status_code!=200 or not r.text:return '' + lines=[] + for ln in r.text.splitlines(): + t=clean(ln) + if not t or t.startswith(('Title:','URL Source:','Published Time:','Markdown Content:','Image:','Description:')):continue + if len(t)>45:lines.append(t) + if sum(len(x) for x in lines)>max_chars:break + return '\n'.join(lines)[:max_chars] + except Exception:return '' + +def _scrape_article_text(url, max_chars=9000): + text=_extract_article_text_bs(url,max_chars) + if len(text)<350:text=_jina_read_text(url,max_chars) + return text + +def _score_relevance(topic, title, text, snippet=''): + keys=[w.lower() for w in re.findall(r'[A-Za-zÀ-ỹ0-9]+',topic) if len(w)>2 and w.lower() not in STOP_WORDS] + hay=(title+' '+snippet+' '+text[:2500]).lower() + if not keys:return 1 + return sum(1 for k in keys if k in hay) + +def _web_research_context(topic): + now=time.time();key=topic.lower().strip() + if key in _TOPIC_CACHE and now-_TOPIC_CACHE[key]['t']<900:return _TOPIC_CACHE[key]['d'] + items=_candidate_urls(topic) + crawled=[] + for it in items: + text=_scrape_article_text(it['url'],9000) + rel=_score_relevance(topic,it.get('title',''),text,it.get('snippet','')) + if text and len(text)>300 and rel>0: + crawled.append({**it,'text':text,'rel':rel}) + elif it.get('snippet') and rel>0: + crawled.append({**it,'text':it['snippet'],'rel':rel,'snippet_only':True}) + crawled=sorted(crawled,key=lambda x:(x.get('rel',0),len(x.get('text',''))),reverse=True)[:6] + blocks=[];sources=[] + for it in crawled: + label='ĐOẠN MÔ TẢ TỪ KẾT QUẢ TÌM KIẾM' if it.get('snippet_only') else 'NỘI DUNG BÀI VIẾT ĐÃ CRAWL' + blocks.append(f"NGUỒN: {it['source']}\nTIÊU ĐỀ: {it['title']}\n{label}:\n{it['text'][:8500]}") + sources.append({'title':it['title'],'url':it['url'],'via':it['source']}) + data={'context':'\n\n---\n\n'.join(blocks),'sources':sources[:8],'count':len(blocks)} + _TOPIC_CACHE[key]={'t':now,'d':data} + return data + +def _topic_image(topic): + try:return f5.base.pollinations_image_url(topic) + except Exception:return 'https://image.pollinations.ai/prompt/'+quote('Vietnamese editorial illustration, '+topic)+'?width=1024&height=576&nologo=true' + +@app.get('/api/topic_sources') +def api_topic_sources(topic:str=Query(...)): + data=_web_research_context(clean(topic)) + return JSONResponse({'count':data.get('count',0),'sources':data.get('sources',[]),'has_context':bool(data.get('context'))}) + +@app.post('/api/topic_post') +async def topic_post_synthesis(request:Request): + body=await request.json();topic=clean(body.get('topic','')) + if not topic:return JSONResponse({'error':'missing topic'},status_code=400) + img=_topic_image(topic);research=_web_research_context(topic);context=research.get('context','');sources=research.get('sources',[]) + if not context or research.get('count',0)==0: + return JSONResponse({'error':'Không tìm/crawl được đủ nội dung về chủ đề này. Hãy thử chủ đề cụ thể hơn hoặc dùng hashtag gợi ý.'},status_code=422) + prompt=f"""Bạn là biên tập viên VNEWS. Người dùng chọn chủ đề: "{topic}". + +Dưới đây là NỘI DUNG các bài viết/đoạn mô tả đã crawl từ internet. Hãy đọc hiểu và TỔNG HỢP thành MỘT BÀI VIẾT HOÀN CHỈNH. Tuyệt đối không bê nguyên văn, không xếp danh sách tiêu đề thành bài viết, không viết kiểu trả lời chat. + +DỮ LIỆU CRAWL: +{context[:30000]} + +Yêu cầu bắt buộc: +- Viết bằng tiếng Việt, văn phong báo điện tử/tạp chí. +- Tiêu đề mới, rõ, hấp dẫn. +- Sapo 2-3 câu nêu vấn đề chính. +- 5-8 đoạn nội dung tổng hợp: bối cảnh, diễn biến/khái niệm, phân tích, tác động, điểm cần lưu ý. +- Dùng thông tin từ nội dung đã crawl để tổng hợp ý; nếu chỉ có mô tả tìm kiếm thì viết thận trọng. +- KHÔNG liệt kê các tiêu đề nguồn. KHÔNG mở đầu bằng "Dưới đây là" hay "Tôi sẽ". +- Cuối bài thêm mục "Nguồn tham khảo" gồm tên nguồn ngắn gọn. +""" + text=await f5.base.qwen_generate(prompt,image_url=img,max_tokens=2800) + if not text or len(text)<500: + parts=[] + for block in context.split('---'): + body=block.split('NỘI DUNG BÀI VIẾT ĐÃ CRAWL:')[-1].split('ĐOẠN MÔ TẢ TỪ KẾT QUẢ TÌM KIẾM:')[-1].strip() + if len(body)>120:parts.append(body) + joined='\n\n'.join(parts)[:8500] + text=(f"{topic}: những điểm chính cần biết\n\n{topic} đang thu hút sự chú ý vì liên quan đến nhiều khía cạnh thực tế. Tổng hợp từ các nội dung thu thập được, có thể nhìn vấn đề qua bối cảnh, tác động và những điểm cần theo dõi.\n\n"+joined+"\n\nNguồn tham khảo: "+', '.join(sorted({s.get('via','') for s in sources if s.get('via')}))) + post=f5.base.make_post(topic,text,img,'','topic_web_synthesis',sources=[s for s in sources if s.get('url')]);post['images']=[img] + posts=f5.base._load_ai_wall();posts.insert(0,post);f5.base._save_ai_wall(posts) + return JSONResponse({'post':post}) + +# ===== Stable newest Dantri/SKDS Shorts ===== +def _yt_ytdlp(handle,count=30): + try: + import yt_dlp + urls=[f'https://www.youtube.com/@{handle}/shorts',f'https://www.youtube.com/@{handle}/videos'] + out=[];seen=set();opts={'quiet':True,'extract_flat':True,'skip_download':True,'playlistend':count,'ignoreerrors':True,'no_warnings':True,'extractor_args':{'youtube':{'player_client':['web']}}} + for url in urls: + with yt_dlp.YoutubeDL(opts) as ydl:info=ydl.extract_info(url,download=False) + for e in (info or {}).get('entries') or []: + vid=e.get('id') or '' + if not re.match(r'^[A-Za-z0-9_-]{11}$',vid) or vid in seen:continue + title=e.get('title') or 'YouTube Short' + if url.endswith('/videos') and '#short' not in title.lower() and 'shorts' not in title.lower():continue + seen.add(vid);out.append({'title':title,'link':'https://www.youtube.com/watch?v='+vid,'img':'https://i.ytimg.com/vi/'+vid+'/hqdefault.jpg','source':'yt','id':vid,'channel':handle}) + if len(out)>=count:break + if len(out)>=count:break + return out + except Exception:return [] +def _yt_html(handle,count=30): + out=[];seen=set() + for suffix in ['shorts','videos']: + try: + r=requests.get(f'https://www.youtube.com/@{handle}/{suffix}',headers=UA,timeout=15);html=r.text + for m in re.finditer(r'"videoId":"([A-Za-z0-9_-]{11})"',html): + vid=m.group(1) + if vid in seen:continue + snip=html[max(0,m.start()-1200):m.start()+2200];title='YouTube Short' + mt=re.search(r'"title":\{"runs":\[\{"text":"([^"]+)"',snip) or re.search(r'"accessibilityText":"([^"]+)"',snip) + if mt:title=clean(mt.group(1).replace('\\n',' ')) + if suffix=='videos' and '#short' not in title.lower() and 'shorts' not in title.lower():continue + seen.add(vid);out.append({'title':title,'link':'https://www.youtube.com/watch?v='+vid,'img':'https://i.ytimg.com/vi/'+vid+'/hqdefault.jpg','source':'yt','id':vid,'channel':handle}) + if len(out)>=count:break + except Exception:pass + if len(out)>=count:break + return out[:count] +def _fallback_shorts(): + try:return f5._fallback_shorts() + except Exception:return [] +@app.get('/api/shorts') +def api_shorts_final6(refresh:int=Query(default=0)): + now=time.time() + if not refresh and _SHORTS_CACHE_FINAL6['d'] and now-_SHORTS_CACHE_FINAL6['t']<600:return JSONResponse(_SHORTS_CACHE_FINAL6['d']) + raw=[] + for h in YOUTUBE_HANDLES:raw.extend(_yt_ytdlp(h,30) or _yt_html(h,30)) + raw.extend(_fallback_shorts()) + seen=set();out=[] + for v in raw: + vid=v.get('id') or '' + if not vid: + m=re.search(r'(?:v=|shorts/|youtu\.be/)([A-Za-z0-9_-]{11})',v.get('link',''));vid=m.group(1) if m else '' + title=_translate_title_vi(v.get('title') or 'YouTube Short');key=vid or re.sub(r'\W+','',title.lower())[:80] + if not key or key in seen:continue + seen.add(key);item=dict(v);item['id']=vid;item['title']=title + if vid:item['link']='https://www.youtube.com/watch?v='+vid;item['img']='https://i.ytimg.com/vi/'+vid+'/hqdefault.jpg' + item['source']='yt';out.append(item) + if len(out)>=40:break + _SHORTS_CACHE_FINAL6.update({'t':now,'d':out}) + return JSONResponse(out) + +FINAL6_INJECT=r''' + + +''' + +@app.get('/') +async def index_final6(): + html=f5.f4.f3.f2.f1._load_index_html() + body=getattr(rt.old,'PATCH_INJECT','')+f5.f4.f3.f2.f1.FINAL_INJECT+f5.f4.f3.FINAL3_INJECT+f5.f4.FINAL4_INJECT+f5.FINAL5_INJECT+FINAL6_INJECT + return HTMLResponse(html.replace('',body+'\n') if '' in html else html+body) + + +# ===== FINAL6B: Vietnam hot hashtags + reliable VN RSS/source retrieval ===== +VN_RSS_FEEDS = [ + ('VnExpress Thời sự','https://vnexpress.net/rss/thoi-su.rss'), + ('VnExpress Thế giới','https://vnexpress.net/rss/the-gioi.rss'), + ('VnExpress Kinh doanh','https://vnexpress.net/rss/kinh-doanh.rss'), + ('VnExpress Công nghệ','https://vnexpress.net/rss/so-hoa.rss'), + ('VnExpress Thể thao','https://vnexpress.net/rss/the-thao.rss'), + ('VnExpress Giải trí','https://vnexpress.net/rss/giai-tri.rss'), + ('VnExpress Sức khỏe','https://vnexpress.net/rss/suc-khoe.rss'), + ('VnExpress Giáo dục','https://vnexpress.net/rss/giao-duc.rss'), + ('Dân trí Xã hội','https://dantri.com.vn/rss/xa-hoi.rss'), + ('Dân trí Thế giới','https://dantri.com.vn/rss/the-gioi.rss'), + ('Dân trí Kinh doanh','https://dantri.com.vn/rss/kinh-doanh.rss'), + ('Dân trí Sức khỏe','https://dantri.com.vn/rss/suc-khoe.rss'), + ('Dân trí Thể thao','https://dantri.com.vn/rss/the-thao.rss'), + ('Dân trí Công nghệ','https://dantri.com.vn/rss/suc-manh-so.rss'), + ('Vietnamnet Thời sự','https://vietnamnet.vn/thoi-su.rss'), + ('Vietnamnet Kinh doanh','https://vietnamnet.vn/kinh-doanh.rss'), + ('Vietnamnet Công nghệ','https://vietnamnet.vn/cong-nghe.rss'), + ('Vietnamnet Thể thao','https://vietnamnet.vn/the-thao.rss'), +] + +def _fetch_rss_items(feed_name, feed_url, max_items=15): + items=[] + try: + r=requests.get(feed_url,headers=UA,timeout=10);r.encoding='utf-8' + soup=BeautifulSoup(r.text,'xml') + for it in soup.find_all('item')[:max_items]: + title=clean(it.find('title').get_text(' ',strip=True) if it.find('title') else '') + link=clean(it.find('link').get_text(strip=True) if it.find('link') else '') + desc=it.find('description').get_text(' ',strip=True) if it.find('description') else '' + desc_txt=clean(BeautifulSoup(desc,'lxml').get_text(' ',strip=True)) + if title and link: + items.append({'title':title,'url':link,'source':feed_name,'snippet':desc_txt}) + except Exception:pass + return items + +def _vn_rss_pool(): + now=time.time();key='vn_rss_pool' + if key in _TOPIC_CACHE and now-_TOPIC_CACHE[key]['t']<600:return _TOPIC_CACHE[key]['d'] + pool=[];seen=set() + for name,url in VN_RSS_FEEDS: + for it in _fetch_rss_items(name,url,12): + if it['url'] not in seen: + seen.add(it['url']);pool.append(it) + _TOPIC_CACHE[key]={'t':now,'d':pool} + return pool + +def _topic_tokens(topic): + toks=[w.lower() for w in re.findall(r'[A-Za-zÀ-ỹ0-9]+',topic or '') if len(w)>1] + return [t for t in toks if t not in STOP_WORDS] + +def _score_topic_item(topic,item): + toks=_topic_tokens(topic) + hay=(item.get('title','')+' '+item.get('snippet','')+' '+item.get('source','')).lower() + if not toks:return 0 + score=0 + for t in toks: + if t in hay:score+=2 if len(t)>3 else 1 + phrase=topic.lower().strip() + if phrase and phrase in hay:score+=8 + return score + +# Override: hashtags must be Việt Nam-focused, using VN news RSS directly. +def _hot_topics(): + now=time.time() + if _HOT_CACHE['d'] and now-_HOT_CACHE['t']<600:return _HOT_CACHE['d'] + pool=_vn_rss_pool() + freq={};display={} + for it in pool[:180]: + title=re.sub(r'\s+-\s+.*$','',it.get('title','')) + # Extract compact Vietnamese hot phrases from current VN headlines. + kws=[] + # quoted/name phrases first + for m in re.findall(r'([A-ZĐÀ-Ỹ][A-Za-zÀ-ỹ0-9]+(?:\s+[A-ZĐÀ-ỸA-Za-zÀ-ỹ0-9][A-Za-zÀ-ỹ0-9]+){1,4})',title): + if len(m)>=6:kws.append(m) + kws += _keywords_from_title(title) + for kw in kws[:5]: + kw=clean(kw) + words=[w for w in kw.split() if w.lower() not in STOP_WORDS] + if len(words)<2:continue + kw=' '.join(words[:5]) + if len(kw)<6 or len(kw)>55:continue + key=kw.lower() + freq[key]=freq.get(key,0)+1 + display[key]=kw + ranked=sorted(freq.items(),key=lambda x:x[1],reverse=True) + topics=[];seen=set() + for key,_ in ranked: + kw=display[key] + if key in seen:continue + seen.add(key) + label='#'+re.sub(r'\s+','',kw.title()) + topics.append({'label':label,'topic':kw}) + if len(topics)>=24:break + # VN fallback, not generic global. + for kw in ['Giá vàng trong nước','Bão và mưa lũ','Bóng đá Việt Nam','Kinh tế Việt Nam','AI tại Việt Nam','Giá xăng dầu','Thị trường chứng khoán Việt Nam','Tuyển Việt Nam','Sức khỏe cộng đồng','An ninh mạng Việt Nam']: + if kw.lower() not in seen:topics.append({'label':'#'+re.sub(r'\s+','',kw.title()),'topic':kw}) + _HOT_CACHE.update({'t':now,'d':topics[:24]}) + return _HOT_CACHE['d'] + +def _candidate_urls(topic): + seen=set();items=[] + # 1) VN RSS pool relevance is most reliable and has direct URLs. + scored=[] + for it in _vn_rss_pool(): + sc=_score_topic_item(topic,it) + if sc>0:scored.append((sc,it)) + for sc,it in sorted(scored,key=lambda x:x[0],reverse=True)[:12]: + if it['url'] not in seen: + seen.add(it['url']);items.append(it) + # 2) Search trusted web if RSS not enough. + queries=[topic+' Việt Nam tin tức',topic+' phân tích Việt Nam',topic+' mới nhất'] + for q in queries: + for it in _ddg_search(q,8): + if it['url'] not in seen: + seen.add(it['url']);items.append(it) + if len(items)>=14:break + # 3) Google News as supplemental titles/direct links. + for it in _google_news_items(topic,10): + if it['url'] not in seen: + seen.add(it['url']);items.append(it) + return items[:24] + +def _web_research_context(topic): + now=time.time();key='ctx2:'+topic.lower().strip() + if key in _TOPIC_CACHE and now-_TOPIC_CACHE[key]['t']<900:return _TOPIC_CACHE[key]['d'] + items=_candidate_urls(topic) + crawled=[] + for it in items: + text=_scrape_article_text(it['url'],9000) + rel=_score_relevance(topic,it.get('title',''),text,it.get('snippet','')) or _score_topic_item(topic,it) + # If RSS item has good snippet, keep it even when full text blocks. + if text and len(text)>300 and rel>0: + crawled.append({**it,'text':text,'rel':rel}) + elif it.get('snippet') and len(it['snippet'])>120 and rel>0: + crawled.append({**it,'text':it['snippet'],'rel':rel,'snippet_only':True}) + crawled=sorted(crawled,key=lambda x:(x.get('rel',0),len(x.get('text',''))),reverse=True)[:7] + blocks=[];sources=[] + for it in crawled: + label='ĐOẠN MÔ TẢ TỪ RSS/TÌM KIẾM' if it.get('snippet_only') else 'NỘI DUNG BÀI VIẾT ĐÃ CRAWL' + blocks.append(f"NGUỒN: {it['source']}\nTIÊU ĐỀ: {it['title']}\n{label}:\n{it['text'][:8500]}") + sources.append({'title':it['title'],'url':it['url'],'via':it['source']}) + data={'context':'\n\n---\n\n'.join(blocks),'sources':sources[:8],'count':len(blocks)} + _TOPIC_CACHE[key]={'t':now,'d':data} + return data + + +# ===== FINAL6C: FAST topic generation (RSS cache first, no slow full-page crawling) ===== +import asyncio +_FAST_TOPIC_CACHE={} +FAST_RSS_FEEDS=[ + ('VnExpress','https://vnexpress.net/rss/tin-moi-nhat.rss'), + ('VnExpress Thời sự','https://vnexpress.net/rss/thoi-su.rss'), + ('VnExpress Thế giới','https://vnexpress.net/rss/the-gioi.rss'), + ('VnExpress Kinh doanh','https://vnexpress.net/rss/kinh-doanh.rss'), + ('VnExpress Công nghệ','https://vnexpress.net/rss/so-hoa.rss'), + ('VnExpress Thể thao','https://vnexpress.net/rss/the-thao.rss'), + ('Dân trí','https://dantri.com.vn/rss/home.rss'), + ('Dân trí Xã hội','https://dantri.com.vn/rss/xa-hoi.rss'), + ('Dân trí Kinh doanh','https://dantri.com.vn/rss/kinh-doanh.rss'), + ('Dân trí Thể thao','https://dantri.com.vn/rss/the-thao.rss'), + ('Dân trí Công nghệ','https://dantri.com.vn/rss/suc-manh-so.rss'), + ('Vietnamnet','https://vietnamnet.vn/rss/tin-moi-nhat.rss'), + ('Vietnamnet Thời sự','https://vietnamnet.vn/thoi-su.rss'), + ('Vietnamnet Kinh doanh','https://vietnamnet.vn/kinh-doanh.rss'), + ('Vietnamnet Công nghệ','https://vietnamnet.vn/cong-nghe.rss'), + ('Vietnamnet Thể thao','https://vietnamnet.vn/the-thao.rss'), +] + +def _fast_fetch_rss(feed_name, feed_url, max_items=20): + items=[] + try: + r=requests.get(feed_url,headers=UA,timeout=6);r.encoding='utf-8' + soup=BeautifulSoup(r.text,'xml') + for it in soup.find_all('item')[:max_items]: + title=clean(it.find('title').get_text(' ',strip=True) if it.find('title') else '') + link=clean(it.find('link').get_text(strip=True) if it.find('link') else '') + desc_raw=it.find('description').get_text(' ',strip=True) if it.find('description') else '' + desc=clean(BeautifulSoup(desc_raw,'lxml').get_text(' ',strip=True)) + if title and link: + items.append({'title':title,'url':link,'source':feed_name,'snippet':desc}) + except Exception:pass + return items + +def _fast_rss_pool(): + now=time.time();key='fast_rss_pool' + if key in _FAST_TOPIC_CACHE and now-_FAST_TOPIC_CACHE[key]['t']<600:return _FAST_TOPIC_CACHE[key]['d'] + pool=[];seen=set() + # Sequential with short timeouts is predictable; RSS is small. + for name,url in FAST_RSS_FEEDS: + for it in _fast_fetch_rss(name,url,16): + if it['url'] not in seen: + seen.add(it['url']);pool.append(it) + _FAST_TOPIC_CACHE[key]={'t':now,'d':pool} + return pool + +def _fast_topic_tokens(topic): + toks=[w.lower() for w in re.findall(r'[A-Za-zÀ-ỹ0-9]+',topic or '') if len(w)>1] + return [t for t in toks if t not in STOP_WORDS] + +def _fast_score(topic,item): + toks=_fast_topic_tokens(topic) + hay=(item.get('title','')+' '+item.get('snippet','')+' '+item.get('source','')).lower() + if not toks:return 0 + score=0 + for t in toks: + if t in hay:score+=3 if len(t)>3 else 1 + phrase=topic.lower().strip() + if phrase and phrase in hay:score+=12 + return score + +def _fast_sources(topic, limit=8): + pool=_fast_rss_pool() + scored=[] + for it in pool: + sc=_fast_score(topic,it) + if sc>0:scored.append((sc,it)) + scored=sorted(scored,key=lambda x:(x[0],len(x[1].get('snippet',''))),reverse=True) + out=[];seen=set() + for sc,it in scored: + if it['url'] in seen:continue + seen.add(it['url']);out.append({**it,'score':sc}) + if len(out)>=limit:break + # If topic too narrow and no match, use top latest from VN RSS as weak context instead of slow crawling. + if not out: + out=pool[:min(limit,8)] + return out + +def _fast_context(topic): + now=time.time();key='fast_ctx:'+topic.lower().strip() + if key in _FAST_TOPIC_CACHE and now-_FAST_TOPIC_CACHE[key]['t']<600:return _FAST_TOPIC_CACHE[key]['d'] + sources=_fast_sources(topic,8) + blocks=[];src=[] + for it in sources: + text=(it.get('snippet') or '').strip() + # Use title + RSS description only: fast and reliable. + blocks.append(f"NGUỒN: {it.get('source','')}\nTIÊU ĐỀ: {it.get('title','')}\nTÓM TẮT RSS:\n{text}") + src.append({'title':it.get('title',''),'url':it.get('url',''),'via':it.get('source','')}) + data={'context':'\n\n---\n\n'.join(blocks),'sources':src,'count':len(blocks)} + _FAST_TOPIC_CACHE[key]={'t':now,'d':data} + return data + +def _fallback_fast_article(topic, sources): + lines=[] + for s in sources[:7]: + title=s.get('title','') + if title:lines.append(title) + body='\n'.join('• '+x for x in lines[:7]) + vias=', '.join(sorted({s.get('via','') for s in sources if s.get('via')})) + return (f"{topic}: những điểm đáng chú ý\n\n" + f"{topic} đang là chủ đề được quan tâm trong dòng tin tức hiện nay. Dựa trên các nguồn tin mới nhất, có thể tổng hợp nhanh một số điểm nổi bật để người đọc nắm bối cảnh và theo dõi tiếp diễn biến.\n\n" + f"Các nguồn tin liên quan cho thấy chủ đề này gắn với những diễn biến sau:\n{body}\n\n" + f"Nhìn chung, đây là vấn đề cần được theo dõi theo nhiều góc độ: bối cảnh, tác động thực tế, phản ứng của các bên liên quan và những thông tin cập nhật tiếp theo. Người đọc nên đối chiếu thêm các nguồn chính thống khi cần quyết định hoặc đánh giá chi tiết.\n\n" + f"Nguồn tham khảo: {vias}") + +# Remove previous slow topic routes and register fast versions last. +app.router.routes=[r for r in app.router.routes if not any(getattr(r,'path',None)==p and m in getattr(r,'methods',set()) for p,m in {('/api/topic_post','POST'),('/api/topic_sources','GET')})] + +@app.get('/api/topic_sources') +def api_topic_sources_fast(topic:str=Query(...)): + data=_fast_context(clean(topic)) + return JSONResponse({'count':data.get('count',0),'sources':data.get('sources',[]),'has_context':bool(data.get('context')),'mode':'fast_rss'}) + +@app.post('/api/topic_post') +async def topic_post_fast(request:Request): + body=await request.json();topic=clean(body.get('topic','')) + if not topic:return JSONResponse({'error':'missing topic'},status_code=400) + img=_topic_image(topic) + research=_fast_context(topic);context=research.get('context','');sources=research.get('sources',[]) + prompt=f"""Bạn là biên tập viên VNEWS. Hãy viết MỘT BÀI VIẾT HOÀN CHỈNH bằng tiếng Việt về chủ đề: {topic} + +Dữ liệu nhanh từ RSS nguồn Việt Nam: +{context[:12000]} + +Yêu cầu: +- Không liệt kê tiêu đề nguồn thành bài viết. +- Tổng hợp thành bài báo/tạp chí hoàn chỉnh. +- Có tiêu đề mới, sapo 2-3 câu, 4-6 đoạn phân tích/bối cảnh/tác động. +- Diễn đạt lại, không sao chép nguyên văn. +- Nếu dữ liệu ít, viết thận trọng và nêu các điểm cần theo dõi. +- Cuối bài có mục Nguồn tham khảo. +""" + text=None + try: + text=await asyncio.wait_for(f5.base.qwen_generate(prompt,image_url=img,max_tokens=1300),timeout=28) + except Exception: + text=None + if not text or len(text)<350: + text=_fallback_fast_article(topic,sources) + post=f5.base.make_post(topic,text,img,'','topic_fast_rss',sources=[s for s in sources if s.get('url')]) + post['images']=[img] + posts=f5.base._load_ai_wall();posts.insert(0,post);f5.base._save_ai_wall(posts) + return JSONResponse({'post':post,'mode':'fast_rss','sources_count':len(sources)}) + + +# ===== FINAL6D: FAST HOME LOAD ===== +_FAST_HOME_CACHE={"t":0,"d":[]} +_FAST_DT_CACHE={"t":0,"d":[]} +_FAST_VNEGO_CACHE={"t":0,"d":[]} +_FAST_HL_CACHE={"t":0,"d":[]} + +def _rss_articles_fast(feed_url, group, source='vne', limit=6): + out=[] + try: + r=requests.get(feed_url,headers=UA,timeout=4);r.encoding='utf-8' + soup=BeautifulSoup(r.text,'xml') + for it in soup.find_all('item')[:limit*2]: + title=clean(it.find('title').get_text(' ',strip=True) if it.find('title') else '') + link=clean(it.find('link').get_text(strip=True) if it.find('link') else '') + desc_raw=it.find('description').get_text(' ',strip=True) if it.find('description') else '' + ds=BeautifulSoup(desc_raw,'lxml') + im=ds.find('img'); img=im.get('src','') if im else '' + desc=clean(ds.get_text(' ',strip=True))[:160] + if title and link: + out.append({'title':title,'link':link,'img':img,'summary':desc,'source':source,'group':group}) + if len(out)>=limit:break + except Exception:pass + return out + +def _fast_homepage(): + now=time.time() + if _FAST_HOME_CACHE['d'] and now-_FAST_HOME_CACHE['t']<600:return _FAST_HOME_CACHE['d'] + feeds=[('Thời Sự','https://vnexpress.net/rss/thoi-su.rss'),('Thế Giới','https://vnexpress.net/rss/the-gioi.rss'),('Kinh Doanh','https://vnexpress.net/rss/kinh-doanh.rss'),('Công Nghệ','https://vnexpress.net/rss/so-hoa.rss'),('Thể Thao','https://vnexpress.net/rss/the-thao.rss'),('Giải Trí','https://vnexpress.net/rss/giai-tri.rss'),('Sức Khỏe','https://vnexpress.net/rss/suc-khoe.rss'),('Giáo Dục','https://vnexpress.net/rss/giao-duc.rss'),('Pháp Luật','https://vnexpress.net/rss/phap-luat.rss'),('Du Lịch','https://vnexpress.net/rss/du-lich.rss')] + arts=[] + try: + from concurrent.futures import ThreadPoolExecutor, as_completed + with ThreadPoolExecutor(max_workers=6) as ex: + futs=[ex.submit(_rss_articles_fast,u,g,'vne',6) for g,u in feeds] + for f in as_completed(futs,timeout=7): + try:arts.extend(f.result() or []) + except Exception:pass + except Exception: + for g,u in feeds[:5]:arts.extend(_rss_articles_fast(u,g,'vne',4)) + if arts:_FAST_HOME_CACHE.update({'t':now,'d':arts}) + return _FAST_HOME_CACHE['d'] or arts + +def _fast_dantri_hot(): + now=time.time() + if _FAST_DT_CACHE['d'] and now-_FAST_DT_CACHE['t']<900:return _FAST_DT_CACHE['d'] + data=_rss_articles_fast('https://dantri.com.vn/rss/home.rss','Tin Nổi Bật','dantri',12) + if data:_FAST_DT_CACHE.update({'t':now,'d':data}) + return data + +def _fast_vnego(): + now=time.time() + if _FAST_VNEGO_CACHE['d'] and now-_FAST_VNEGO_CACHE['t']<900:return _FAST_VNEGO_CACHE['d'] + out=[] + try: + r=requests.get('https://vnexpress.net/vne-go',headers=UA,timeout=4);r.encoding='utf-8' + soup=BeautifulSoup(r.text,'lxml');seen=set() + for a in soup.find_all('a',href=True): + href=a.get('href','');title=clean(a.get('title','') or a.get_text(' ',strip=True)) + if not title or len(title)<8 or not href.startswith('http') or href in seen:continue + if '/vne-go' not in href and '/video/' not in href:continue + seen.add(href);img='';im=a.find('img') or (a.parent.find('img') if a.parent else None) + if im:img=im.get('data-src') or im.get('src','') + out.append({'title':title,'link':href,'img':img,'source':'vne-video'}) + if len(out)>=10:break + except Exception:pass + _FAST_VNEGO_CACHE.update({'t':now,'d':out}) + return out + +def _fast_highlights(): + now=time.time() + if _FAST_HL_CACHE['d'] and now-_FAST_HL_CACHE['t']<900:return _FAST_HL_CACHE['d'] + _FAST_HL_CACHE.update({'t':now,'d':[]}) + return [] + +for _p in ['/api/homepage','/api/dantri_hot','/api/vne_video','/api/highlights']: + app.router.routes=[r for r in app.router.routes if not (getattr(r,'path',None)==_p and 'GET' in getattr(r,'methods',set()))] +@app.get('/api/homepage') +def api_homepage_fast():return JSONResponse(_fast_homepage()) +@app.get('/api/dantri_hot') +def api_dantri_hot_fast():return JSONResponse(_fast_dantri_hot()) +@app.get('/api/vne_video') +def api_vne_video_fast():return JSONResponse(_fast_vnego()) +@app.get('/api/highlights') +def api_highlights_fast():return JSONResponse(_fast_highlights()) + +FINAL6_FAST_HOME_INJECT = """ + +""" +app.router.routes=[r for r in app.router.routes if not (getattr(r,'path',None)=='/' and 'GET' in getattr(r,'methods',set()))] +@app.get('/') +async def index_final6_fast_home(): + html=f5.f4.f3.f2.f1._load_index_html() + body=getattr(rt.old,'PATCH_INJECT','')+f5.f4.f3.f2.f1.FINAL_INJECT+f5.f4.f3.FINAL3_INJECT+f5.f4.FINAL4_INJECT+f5.FINAL5_INJECT+FINAL6_INJECT+FINAL6_FAST_HOME_INJECT + return HTMLResponse(html.replace('',body+'\n') if '' in html else html+body) + + +# ===== FINAL6E: SHOW SOURCE CONTENTS IN TOPIC ARTICLE ===== +def _extract_source_details_from_context(context, sources): + details=[] + # Map source urls by title for URL/via enrichment + src_by_title={clean(s.get('title','')):s for s in (sources or [])} + for block in (context or '').split('---'): + block=block.strip() + if not block:continue + via='';title='';content='' + m=re.search(r'NGUỒN:\s*(.*)',block) + if m:via=clean(m.group(1)) + m=re.search(r'TIÊU ĐỀ:\s*(.*)',block) + if m:title=clean(m.group(1)) + if 'NỘI DUNG BÀI VIẾT ĐÃ CRAWL:' in block: + content=block.split('NỘI DUNG BÀI VIẾT ĐÃ CRAWL:',1)[1] + elif 'TÓM TẮT RSS:' in block: + content=block.split('TÓM TẮT RSS:',1)[1] + elif 'ĐOẠN MÔ TẢ' in block: + content=re.split(r'ĐOẠN MÔ TẢ[^:]*:',block,1)[-1] + content=clean(content) + if not title and not content:continue + s=src_by_title.get(title,{}) + details.append({'title':title or s.get('title','Nguồn tham khảo'),'url':s.get('url',''),'via':via or s.get('via',''),'content':content[:1800]}) + if len(details)>=8:break + return details + +# Remove prior topic endpoint and register one that stores source_details in post. +app.router.routes=[r for r in app.router.routes if not (getattr(r,'path',None)=='/api/topic_post' and 'POST' in getattr(r,'methods',set()))] + +@app.post('/api/topic_post') +async def topic_post_with_source_contents(request:Request): + body=await request.json();topic=clean(body.get('topic','')) + if not topic:return JSONResponse({'error':'missing topic'},status_code=400) + img=_topic_image(topic) + research=_fast_context(topic) if '_fast_context' in globals() else _web_research_context(topic) + context=research.get('context','');sources=research.get('sources',[]) + details=_extract_source_details_from_context(context,sources) + if not context or not details: + return JSONResponse({'error':'Không tìm/crawl được đủ nội dung về chủ đề này. Hãy thử chủ đề cụ thể hơn hoặc dùng hashtag gợi ý.'},status_code=422) + source_brief='\n\n'.join([f"[{i+1}] {d.get('title','')} ({d.get('via','')})\n{d.get('content','')[:1400]}" for i,d in enumerate(details)]) + prompt=f"""Bạn là biên tập viên VNEWS. Hãy viết MỘT BÀI VIẾT HOÀN CHỈNH bằng tiếng Việt về chủ đề: {topic} + +Dưới đây là nội dung từng nguồn đã thu thập. Hãy tổng hợp ý chính, không sao chép nguyên văn, không biến các tiêu đề thành danh sách. + +NỘI DUNG NGUỒN: +{source_brief[:18000]} + +Yêu cầu: +- Tiêu đề mới, rõ, hấp dẫn. +- Sapo 2-3 câu. +- 5-8 đoạn phân tích/bối cảnh/tác động/điểm cần lưu ý. +- Không dùng câu "Dưới đây là" hoặc "Tôi sẽ". +- Cuối bài có mục "Nguồn tham khảo" nêu tên nguồn. +""" + text=None + try: + import asyncio + text=await asyncio.wait_for(f5.base.qwen_generate(prompt,image_url=img,max_tokens=1700),timeout=35) + except Exception: + text=None + if not text or len(text)<350: + bullets='\n'.join([f"• {d['title']}: {d.get('content','')[:320]}" for d in details[:6]]) + vias=', '.join(sorted({d.get('via','') for d in details if d.get('via')})) + text=(f"{topic}: tổng hợp những điểm đáng chú ý\n\n" + f"{topic} đang được nhiều nguồn tin đề cập với các góc nhìn khác nhau. Dưới đây là phần tổng hợp nhanh từ những nội dung đã thu thập được.\n\n" + f"{bullets}\n\n" + f"Nhìn chung, chủ đề này cần được theo dõi thêm ở các khía cạnh: bối cảnh, tác động thực tế, phản ứng của các bên liên quan và các diễn biến mới trong thời gian tới.\n\n" + f"Nguồn tham khảo: {vias}") + post=f5.base.make_post(topic,text,img,'','topic_fast_rss_with_sources',sources=[s for s in sources if s.get('url')]) + post['images']=[img] + post['source_details']=details + posts=f5.base._load_ai_wall();posts.insert(0,post);f5.base._save_ai_wall(posts) + return JSONResponse({'post':post,'mode':'fast_rss_with_source_details','sources_count':len(details)}) + +FINAL6E_INJECT = """ + + +''' diff --git a/ai_runtime_fix.py b/ai_runtime_fix.py new file mode 100644 index 0000000000000000000000000000000000000000..0ec96f7c3c57d95b682765689d6ed0964058cd85 --- /dev/null +++ b/ai_runtime_fix.py @@ -0,0 +1,394 @@ +"""VNEWS Short Video Fix - standalone module with clean registration. +This module MUST be imported LAST to register /api/ai/short endpoints. +FIX v1: No route filtering issues - registers endpoints unconditionally. +FIX v2: SSE inline endpoint for auto homepage updates +""" +import os +import re +import time +import json +import sys +import logging +import asyncio +import hashlib +import subprocess +import requests +from datetime import datetime, timezone, timedelta +from urllib.parse import urlparse +from fastapi import Request, Query +from fastapi.responses import JSONResponse, FileResponse + +# Import dependencies +try: + import ai_ext as base +except ImportError: + import ai_runtime_final6 as base + +# Try to import app from various sources +try: + from app_v2_entry import app +except ImportError: + try: + from main import app + except ImportError: + from ai_runtime_final6 import app + +_log = logging.getLogger("short_fix") +_log.setLevel(logging.INFO) +if not _log.handlers: + _log.addHandler(logging.StreamHandler(sys.stderr)) + +DATA_DIR = "/data" if os.path.isdir("/data") else "/app/data" +os.makedirs(DATA_DIR, exist_ok=True) +SHORTS_DIR = os.path.join(DATA_DIR, "ai_shorts") +os.makedirs(SHORTS_DIR, exist_ok=True) + +# ===== VIETNAMESE FONT DETECTION ===== +_VN_FONT_REG = None +_VN_FONT_BOLD = None + +def _get_vn_fonts(): + """Find Vietnamese-supporting fonts.""" + global _VN_FONT_REG, _VN_FONT_BOLD + if _VN_FONT_REG is not None: + return _VN_FONT_REG, _VN_FONT_BOLD + + try: + from PIL import ImageFont + except Exception: + _log.error("PIL not available!") + return None, None + + # Priority: Noto > DejaVu > Liberation + reg_paths = [ + "/usr/share/fonts/truetype/noto/NotoSans-Regular.ttf", + "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf", + "/usr/share/fonts/truetype/liberation/LiberationSans-Regular.ttf", + "/usr/share/fonts/truetype/freefont/FreeSans.ttf", + ] + bold_paths = [ + "/usr/share/fonts/truetype/noto/NotoSans-Bold.ttf", + "/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf", + "/usr/share/fonts/truetype/liberation/LiberationSans-Bold.ttf", + "/usr/share/fonts/truetype/freefont/FreeSans.ttf", + ] + + for path in reg_paths: + if os.path.exists(path): + try: + _VN_FONT_REG = ImageFont.truetype(path, 40) + _log.info(f"Found regular font: {path}") + break + except: + continue + + for path in bold_paths: + if os.path.exists(path): + try: + _VN_FONT_BOLD = ImageFont.truetype(path, 52) + _log.info(f"Found bold font: {path}") + break + except: + continue + + if _VN_FONT_REG is None: + _VN_FONT_REG = ImageFont.load_default() + if _VN_FONT_BOLD is None: + _VN_FONT_BOLD = _VN_FONT_REG + + return _VN_FONT_REG, _VN_FONT_BOLD + + +def _clean(s): + import html as html_lib + return re.sub(r"\s+", " ", html_lib.unescape(str(s or ""))).strip() + + +# ===== ROBUST TEXT SEGMENTATION ===== +def _split_into_segments(text, max_segments=10, min_len=30): + """Split text into segments - multi strategy.""" + text = _clean(text) + if not text: + return [] + + # Strategy 1: bullet points + lines = text.split('\n') + segmented = [] + for line in lines: + line = _clean(line) + line_bare = re.sub(r'^[•\-\*\d\.\)\s]+', '', line).strip() + if len(line_bare) > min_len: + segmented.append(line_bare) + elif len(line) > min_len: + segmented.append(line) + + # Strategy 2: sentences (Vietnamese) + if len(segmented) < 2: + sents = re.split(r'(?<=[.!?])\s+(?=[A-Z0-9À-ỸĐ])', text) + segmented = [s for s in sents if len(_clean(s)) > min_len] + + # Strategy 3: character chunks + if not segmented: + words = text.split() + for i in range(0, min(len(words), max_segments * 20), 20): + chunk = ' '.join(words[i:i+20]) + if len(chunk) > min_len: + segmented.append(chunk) + + # Strategy 4: fallback + if not segmented: + segmented = [text[:300]] + + return segmented[:max_segments] + + +# ===== SHORT VIDEO GENERATOR ===== +def _gen_short_core(post, work_dir): + """Core short generation - returns video path or None.""" + post_id = post.get('id', '') + text = post.get('text', '') or post.get('title', '') + + if not post_id or len(text) < 100: + _log.error(f"Invalid post: id={post_id}, text_len={len(text)}") + return None + + segments = _split_into_segments(text, max_segments=10, min_len=30) + if not segments: + _log.error("No segments generated") + return None + + _log.info(f"Generating short: {len(segments)} segments") + + seg_hash = hashlib.md5(('|'.join(segments) + 'nu').encode()).hexdigest()[:8] + suffix = f"_nu_{seg_hash}" + out_mp4 = os.path.join(work_dir, f"{post_id}{suffix}.mp4") + + if os.path.exists(out_mp4): + _log.info(f"Already exists: {out_mp4}") + return out_mp4 + + # Check dependencies + try: + subprocess.run(['ffmpeg', '-version'], capture_output=True, timeout=5) + except Exception as e: + _log.error(f"ffmpeg missing: {e}") + return None + + # Download image + img_path = os.path.join(work_dir, 'bg.jpg') + downloaded = False + try: + img_url = post.get('img', '') + if img_url and img_url.startswith('http'): + r = requests.get(img_url, headers={'User-Agent': 'Mozilla/5.0'}, timeout=12) + if r.status_code == 200: + with open(img_path, 'wb') as f: + f.write(r.content) + downloaded = True + except Exception as e: + _log.warning(f"Image download: {e}") + + try: + from PIL import Image, ImageDraw + has_pil = True + except: + has_pil = False + _log.warning("PIL not available") + + try: + from gtts import gTTS + has_tts = True + except: + has_tts = False + _log.warning("gTTS not available") + + parts = [] + + for i, seg in enumerate(segments[:10]): + frame = os.path.join(work_dir, f'frame_{i}.jpg') + audio = os.path.join(work_dir, f'audio_{i}.mp3') + part = os.path.join(work_dir, f'part_{i}.mp4') + + # Create frame + try: + if has_pil: + _make_frame(post, seg, img_path, downloaded, frame) + else: + subprocess.run(['ffmpeg', '-y', '-f', 'lavfi', '-i', + 'color=c=black:s=1080x1920:d=1', '-frames:v', '1', frame], + capture_output=True, timeout=20) + except Exception as e: + _log.error(f"Frame error: {e}") + continue + + # Create audio + if has_tts: + try: + tts = _clean(seg)[:300] + gTTS(tts, lang='vi', slow=False).save(audio) + except Exception as e: + _log.warning(f"TTS error: {e}") + audio = None + + # Combine + dur = 10 + try: + cmd = ['ffmpeg', '-y', '-loop', '1', '-t', str(dur), '-i', frame] + if has_tts and os.path.exists(audio): + cmd += ['-i', audio, '-shortest'] + else: + cmd += ['-f', 'lavfi', '-i', 'anullsrc', '-shortest'] + cmd += ['-c:v', 'libx264', '-tune', 'stillimage', '-pix_fmt', 'yuv420p', + '-c:a', 'aac', '-b:a', '128k', part] + subprocess.run(cmd, capture_output=True, timeout=120) + if os.path.exists(part) and os.path.getsize(part) > 5000: + parts.append(part) + except Exception as e: + _log.error(f"Part combine error: {e}") + + if not parts: + _log.error("No video parts created!") + return None + + # Concatenate + try: + concat = os.path.join(work_dir, 'list.txt') + with open(concat, 'w') as f: + for p in parts: + f.write(f"file '{p}'\n") + subprocess.run(['ffmpeg', '-y', '-f', 'concat', '-safe', '0', '-i', concat, '-c', 'copy', out_mp4], + capture_output=True, timeout=180) + _log.info(f"Short created: {out_mp4}") + return out_mp4 + except Exception as e: + _log.error(f"Concat error: {e}") + return None + + +def _make_frame(post, text, img_path, downloaded, out_path): + """Create video frame with Vietnamese font.""" + from PIL import Image, ImageDraw + _get_vn_fonts() + + W, H = 1080, 1920 + bg = Image.new('RGB', (W, H), (15, 23, 38)) + d = ImageDraw.Draw(bg) + + # Background image + if downloaded and os.path.exists(img_path): + try: + im = Image.open(img_path).convert('RGB') + im = im.resize((W, 760)) + bg.paste(im, (0, 0)) + except: + pass + + # Title + d.rectangle([0, 0, W, 100], fill=(25, 118, 210)) + ttl = post.get('title', '')[:50] + if _VN_FONT_BOLD: + d.text((W//2, 50), ttl, fill='white', font=_VN_FONT_BOLD, anchor='mm') + + # Content + y = 150 + for ln in _wrap_text(d, text[:200], _VN_FONT_REG, 80, 920, 10): + d.text((80, y), ln, fill='white', font=_VN_FONT_REG) + y += 55 + + bg.save(out_path, quality=85) + + +def _wrap_text(draw, text, font, x, max_w, max_lines): + """Word wrap text.""" + words = text.split() + lines = [] + cur = [] + for w in words: + test = ' '.join(cur + [w]) + try: + w_px = draw.textbbox((0, 0), test, font=font)[2] + except: + w_px = len(test) * 22 + if w_px <= max_w: + cur.append(w) + else: + if cur: + lines.append(' '.join(cur)) + cur = [w] + if len(lines) >= max_lines: + break + if cur and len(lines) < max_lines: + lines.append(' '.join(cur)) + return lines + + +def _gen_short_sync(post) -> str: + """Sync wrapper - returns video URL.""" + work = os.path.join(SHORTS_DIR, f"work_{post.get('id', int(time.time()))}") + os.makedirs(work, exist_ok=True) + result = _gen_short_core(post, work) + if result: + # Update wall + try: + wall = base._load_ai_wall() + for i, p in enumerate(wall): + if str(p.get('id')) == str(post.get('id')): + p['video'] = f'/api/ai/short-file/{post.get("id")}_nu_{hashlib.md5(str(post).encode()).hexdigest()[:8]}' + wall[i] = p + break + base._save_ai_wall(wall) + # Notify SSE for auto-update + try: + from auto_update_sse import notify_new_short + notify_new_short(post) + except: + pass + except Exception as e: + _log.warning(f"Wall update: {e}") + return result + return '' + + +# ===== REGISTER ENDPOINTS - MUST BE AT MODULE LEVEL ===== +@app.post('/api/ai/short/{post_id}') +async def api_short_generate(post_id: str, request: Request): + _log.info(f"POST /api/ai/short/{post_id}") + wall = base._load_ai_wall() + post = next((p for p in wall if str(p.get('id')) == str(post_id)), None) + if not post: + return JSONResponse({'error': 'Post not found in wall'}, status_code=404) + + if post.get('video'): + return JSONResponse({'post': post, 'video': post['video'], 'status': 'done'}) + + loop = asyncio.get_event_loop() + result = await loop.run_in_executor(None, _gen_short_sync, post) + + if result: + # Get the video URL from wall (updated in _gen_short_sync) + wall = base._load_ai_wall() + post = next((p for p in wall if str(p.get('id')) == str(post_id)), post) + return JSONResponse({'post': post, 'video': post.get('video'), 'status': 'done'}) + return JSONResponse({'error': 'Video generation failed'}, status_code=500) + + +@app.get('/api/ai/short-file/{file_id:path}') +async def api_short_file(file_id: str): + safe = re.sub(r'[^\w\-.]', '_', file_id)[:100] + for fname in os.listdir(SHORTS_DIR) if os.path.isdir(SHORTS_DIR) else []: + if fname.endswith('.mp4') and safe in fname: + return FileResponse(os.path.join(SHORTS_DIR, fname), media_type='video/mp4') + return JSONResponse({'error': 'Not found'}, status_code=404) + + +# ===== SSE ENDPOINT FOR AUTO-UPDATE ===== +try: + from auto_update_sse import sse_events as _sse_handler + app.add_api_route('/api/events', _sse_handler, methods=['GET']) + _log.info("SSE endpoint registered at /api/events") +except Exception as e: + _log.warning(f"SSE route not loaded: {e}") + + +# Log startup +_log.info("Short video endpoints registered") \ No newline at end of file diff --git a/ai_runtime_patch_fast.py b/ai_runtime_patch_fast.py new file mode 100644 index 0000000000000000000000000000000000000000..126cc9a14012fabb8b4f72064810a56a1c663802 --- /dev/null +++ b/ai_runtime_patch_fast.py @@ -0,0 +1,188 @@ +"""Final patch v2: fix topic rewrite, remove duplicate short slide, full short interaction buttons.""" +import re, threading, time, json, os, asyncio +import ai_runtime_final6 as f6 +from ai_runtime_final6 import app, rt, f5, HTMLResponse, JSONResponse, Request, Query +import html as html_lib +from urllib.parse import urlparse + +def clean(s):return re.sub(r"\s+"," ",html_lib.unescape(str(s or ""))).strip() +def _domain(u): + try:return urlparse(u or '').netloc.replace('www.','') + except:return '' +DATA_DIR="/data" if os.path.isdir('/data') else "/app/data" +os.makedirs(DATA_DIR,exist_ok=True) +SHORT_COMMENTS_FILE=os.path.join(DATA_DIR,'short_comments.json') +TTL_24H=86400;HAS_PERSISTENT=os.path.isdir('/data') +def _lj(p,d): + try: + if os.path.exists(p):return json.load(open(p,'r',encoding='utf-8')) + except:pass + return d +def _sj(p,d): + try:os.makedirs(os.path.dirname(p),exist_ok=True);open(p+'.tmp','w',encoding='utf-8').write(json.dumps(d,ensure_ascii=False));os.replace(p+'.tmp',p) + except:pass +def _cleanup(): + n=int(time.time());ps=f5.base._load_ai_wall();f=[p for p in ps if n-int(p.get('ts') or 0)300 else None);return JSONResponse(_bg_home['d']) + if hasattr(f6,'_fast_homepage'):d=f6._fast_homepage();_bg_home.update({"t":n,"d":d or []});return JSONResponse(d or []) + return JSONResponse([]) +@app.get('/api/shorts') +def _sh(refresh:int=Query(default=0)): + n=time.time() + if _bg_shorts['d'] and (not refresh or n-_bg_shorts['t']<120):(threading.Thread(target=_bg,daemon=True).start() if n-_bg_shorts['t']>600 else None);return JSONResponse(_bg_shorts['d']) + return f6.api_shorts_final6(refresh=refresh) if hasattr(f6,'api_shorts_final6') else JSONResponse([]) +@app.get('/api/ai_wall') +def _w():n=int(time.time());return JSONResponse({'posts':[p for p in f5.base._load_ai_wall() if n-int(p.get('ts') or 0)150 else None) + ac='\n---\n'.join(parts) if parts else (p.get('text') or '') + title=p.get('title','') + text=None + try:text=await asyncio.wait_for(f5.base.qwen_generate(f'Viết lại:\nChủ đề: {title}\n{ac[:16000]}\n\nTiêu đề mới + 4-6 ý + nguồn.',image_url=p.get('img'),max_tokens=1200),timeout=35) + except:pass + if not text or len(text)<100:text=f"Tóm tắt: {title}\n\n{ac[:1500]}\n\nNguồn: VNEWS AI" + np=f5.base.make_post('Rewrite: '+title,text,p.get('img',''),'','rewrite_topic',sources=p.get('sources',[]));np['images']=p.get('images',[]) + all_p=f5.base._load_ai_wall();all_p.insert(0,np);f5.base._save_ai_wall(all_p);return JSONResponse({'post':np}) +@app.post('/api/topic_post') +async def _tp(request:Request): + b=await request.json();topic=clean(b.get('topic','')) + if not topic:return JSONResponse({'error':'missing topic'},status_code=400) + img=f6._topic_image(topic);research=f6._fast_context(topic) if hasattr(f6,'_fast_context') else f6._web_research_context(topic) + ctx=research.get('context','');src=research.get('sources',[]);det=f6._extract_source_details_from_context(ctx,src) if hasattr(f6,'_extract_source_details_from_context') else [] + if not ctx or not src:return JSONResponse({'error':'Không tìm được nội dung.'},status_code=422) + sb='\n\n'.join([f"[{i+1}] {d.get('title','')} ({d.get('via','')})\n{d.get('content','')[:1400]}" for i,d in enumerate(det)]) if det else ctx[:18000] + text=None + try:text=await asyncio.wait_for(f5.base.qwen_generate(f'Viết bài tiếng Việt VỀ: "{topic}"\nNGUỒN:\n{sb[:18000]}\nCHỈ viết về "{topic}". 5-8 đoạn. Cuối có nguồn.',image_url=img,max_tokens=1700),timeout=35) + except:pass + if not text or len(text)<300:text=f"{topic}: tổng hợp\n\n"+'\n'.join([f"• {d['title']}: {d.get('content','')[:300]}" for d in (det or [])[:6]])+"\n\nNguồn: "+', '.join(sorted({d.get('via','') for d in (det or []) if d.get('via')})) + post=f5.base.make_post(topic,text,img,'','topic_focused',sources=[s for s in src if s.get('url')]);post['images']=[img];post['source_details']=det + ps=f5.base._load_ai_wall();ps.insert(0,post);f5.base._save_ai_wall(ps);return JSONResponse({'post':post}) + +PATCH_INJECT=r''' + +
+ +''' + +@app.get('/') +async def _index(): + html=f5.f4.f3.f2.f1._load_index_html() + body=getattr(rt.old,'PATCH_INJECT','')+f5.f4.f3.f2.f1.FINAL_INJECT+f5.f4.f3.FINAL3_INJECT+f5.f4.FINAL4_INJECT+f5.FINAL5_INJECT + body+=getattr(f6,'FINAL6_INJECT','');body+=getattr(f6,'FINAL6_FAST_HOME_INJECT','');body+=getattr(f6,'FINAL6E_INJECT','') + body+=PATCH_INJECT + return HTMLResponse(html.replace('',body+'\n') if '' in html else html+body) diff --git a/ai_runtime_patch_final.py b/ai_runtime_patch_final.py new file mode 100644 index 0000000000000000000000000000000000000000..0512ad0ef58446a2f791f8f85e0457950cdd54d3 --- /dev/null +++ b/ai_runtime_patch_final.py @@ -0,0 +1,78 @@ +"""Final patch: homepage fix + AI topics at top + SSE auto-update""" +import re, json, time +from fastapi.responses import HTMLResponse, JSONResponse +from fastapi import Query + +# Import chain - must be after ai_runtime_final6 +try: + import ai_runtime_final6 as f6 + from ai_runtime_final6 import app, f5 + from main import rt +except Exception as e: + print(f"[ERROR] f6 import: {e}") + f6 = None + f5 = None + rt = None + +PATCH_CSS_JS = r''' + +
+ +''' + +# Register route if possible +if f6 and app: + try: + # Remove duplicate / route to avoid conflict + original_routes = [r for r in app.router.routes if not (getattr(r,'path',None)=='/' and 'GET' in getattr(r,'methods',set()))] + app.router.routes = original_routes + + @app.get('/') + async def patch_homepage(): + html = f5.f4.f3.f2.f1._load_index_html() if f5 else "" + body = "" + if hasattr(rt,'old') and hasattr(rt.old,'PATCH_INJECT'): + body += getattr(rt.old,'PATCH_INJECT','') + if f5: + body += getattr(f5.f4.f3.f2.f1,'FINAL_INJECT','') if hasattr(f5,'f4') else '' + body += getattr(f5.f4.f3,'FINAL3_INJECT','') if hasattr(f5,'f4') else '' + body += getattr(f5.f4,'FINAL4_INJECT','') if hasattr(f5,'f4') else '' + body += getattr(f5,'FINAL5_INJECT','') if hasattr(f5,'f4') else '' + body += getattr(f6,'FINAL6_INJECT','') if f6 else '' + body += getattr(f6,'FINAL6_FAST_HOME_INJECT','') if f6 else '' + body += getattr(f6,'FINAL6E_INJECT','') if f6 else '' + body += PATCH_CSS_JS + if '' in html: + html = html.replace('', body + '\n') + else: + html = html + body + return HTMLResponse(html) + except Exception as e: + print(f"[ERROR] register route: {e}") \ No newline at end of file diff --git a/ai_short_v2.py b/ai_short_v2.py new file mode 100644 index 0000000000000000000000000000000000000000..482e09996007086032db92ba5c6ff2a648a670d7 --- /dev/null +++ b/ai_short_v2.py @@ -0,0 +1,1691 @@ +"""VNEWS Short AI v2 — Short creator with TikTok background music, uploaded audio, +uploaded video/image background, and 'recreate from designed slides' (image-only, +reusing previous short audio, NO text overlays) mode. + +Registered by app_v2_entry.py (import ai_short_v2). Uses the SAME wall store +(WALL_FILE) as the wall endpoints so posts created by the designer survive and +can be re-shorted purely from designed images. +""" +import os +import re +import json +import time +import uuid +import hashlib +import subprocess +import threading + +import requests +from urllib.parse import quote as urllib_quote +from fastapi import Request, UploadFile, File, Query +from fastapi.responses import JSONResponse, FileResponse, Response, StreamingResponse + +try: + import yt_dlp +except Exception: # pragma: no cover + yt_dlp = None + +try: + from PIL import Image, ImageDraw, ImageFont +except Exception: # pragma: no cover + Image = ImageDraw = ImageFont = None + +try: + from main import app +except Exception: # pragma: no cover + from fastapi import FastAPI + app = FastAPI() + +DATA_DIR = "/data" if os.path.isdir("/data") else os.path.join(os.path.dirname(os.path.abspath(__file__)), "data") +WALL_FILE = os.path.join(DATA_DIR, "wall_posts.json") +SHORTS_DIR = os.path.join(DATA_DIR, "ai_shorts") +UPLOAD_DIR = os.path.join(DATA_DIR, "short_uploads") +WALL_IMG_DIR = os.path.join(DATA_DIR, "wall_imgs") +os.makedirs(SHORTS_DIR, exist_ok=True) +os.makedirs(UPLOAD_DIR, exist_ok=True) + +_wl_lock = threading.Lock() + +UA_HEADERS = { + "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36", + "Accept-Language": "vi-VN,vi;q=0.9,en;q=0.8", +} + +# --------------------------------------------------------------------------- +# Video scraping (YouTube / TikTok / news sites) via yt-dlp + oEmbed fallback +# --------------------------------------------------------------------------- +_SCRAPE_CACHE = {} # url -> {ok, ...} (in-memory, TTL 30 min) +_SCRAPE_TTL = 30 * 60 + +YTDLP_OPTS = { + "format": "best[height<=720][ext=mp4]/best[height<=720]/best", + "quiet": True, + "no_warnings": True, + "noplaylist": True, + "socket_timeout": 15, + "retries": 3, + "ignoreerrors": False, + "extractor_args": {"youtube": {"player_client": ["tv", "ios", "mweb"]}}, + "http_headers": UA_HEADERS, +} + + +# --------------------------------------------------------------------------- +# YouTube cookies (bypasses the "Sign in to confirm you're not a bot" block). +# Sources, in priority order: +# 1. YT_COOKIES env var (HF Space secret — raw Netscape-format cookies.txt) +# 2. /app/cookies.txt (a cookies.txt file baked into the repo/runtime) +# yt-dlp reads a Netscape cookies.txt via the 'cookiefile' option. +# --------------------------------------------------------------------------- +_cookie_file = None + + +def _ensure_cookies(): + """Materialise a cookies.txt (Netscape format) for yt-dlp if any cookie + source is available. Returns the cookiefile path or None.""" + global _cookie_file + if _cookie_file and os.path.exists(_cookie_file): + return _cookie_file + content = None + # 1. HF Space secret YT_COOKIES (raw cookies.txt content) + secret = os.environ.get("YT_COOKIES", "").strip() + if secret and ("# Netscape" in secret or "#HTTP" in secret or "youtube.com" in secret or "\tTRUE" in secret): + content = secret + # 2. baked-in file + if not content: + for p in ("/app/cookies.txt", "cookies.txt"): + if os.path.exists(p): + content = open(p, encoding="utf-8", errors="ignore").read() + break + if not content: + _cookie_file = None + return None + try: + path = os.path.join(DATA_DIR, "yt_cookies.txt") + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "w", encoding="utf-8") as f: + f.write(content) + _cookie_file = path + return path + except Exception: + return None + + +def _ytdlp_opts(): + """YTDLP opts + cookiefile (only when cookies are available).""" + opts = dict(YTDLP_OPTS) + cf = _ensure_cookies() + if cf: + opts["cookiefile"] = cf + # also add "cookiesfrombrowser" fallback? No — datacenter IPs can't read + # a local browser. cookiefile is the reliable path. + return opts + + +def _oembed_probe(url): + """Cheap metadata fallback for TikTok/Facebook/Instagram (no direct URL).""" + try: + host = (re.sub(r"^https?://", "", url).split("/")[0] or "").lower() + api = None + if "tiktok" in host: + api = "https://www.tiktok.com/oembed?url=" + urllib_quote(url, safe="") + elif "facebook" in host or "fb.watch" in host: + api = "https://www.facebook.com/plugins/video/oembed.json?url=" + urllib_quote(url, safe="") + if not api: + return None + r = requests.get(api, headers=UA_HEADERS, timeout=12) + if r.status_code != 200: + return None + j = r.json() + return { + "ok": True, + "title": (j.get("title") or "").strip()[:200], + "thumbnail": (j.get("thumbnail_url") or "").strip(), + "duration": None, + "extractor": "oembed", + "direct_url": None, + "previewable": False, # no direct streamable URL from oEmbed alone + "note": "Có thể lấy được thông tin, nhưng không tải được video trực tiếp từ link này.", + } + except Exception: + return None + + +def _is_real_http(u): + """True if a candidate URL is a real http(s) media URL (not a JS template).""" + u = (u or "").strip() + if not re.match(r"^https?://", u): + return False + if "'" in u or '"' in u or "+" in u or u.count("(") > 0: + return False + return True + + +def _scrape_page_html(url): + """Fallback scraper for news sites (24h, dantri, znews...): parse the article + HTML for og:video /