File size: 18,278 Bytes
bec08c0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
"""Final runtime overrides for VNEWS AI UI, article-only images, shareable AI wall, and robust Vietnamese shorts."""
import os, re, requests, subprocess, time
from urllib.parse import urlparse, quote
import ai_runtime as rt
from ai_runtime import app
import ai_ext as base
from fastapi import Request, Query
from fastapi.responses import HTMLResponse, JSONResponse, FileResponse
try:
    from PIL import Image, ImageDraw, ImageFont
except Exception:
    Image = ImageDraw = ImageFont = None

RESTORE_INDEX_URL = "https://huggingface.co/spaces/bep40/vnews/raw/restore-33c3dda/static/index.html"
SPACE_URL = "https://bep40-vnews.hf.space"
DEFAULT_IMG = "https://s1.vnecdn.net/vnexpress/restruct/i/v9505/logo_default.jpg"

# Only voices that support Vietnamese reliably. Extra labels map to these Vietnamese neural voices.
VN_VOICES = {
    "nu": "vi-VN-HoaiMyNeural", "female": "vi-VN-HoaiMyNeural", "hoaimy": "vi-VN-HoaiMyNeural",
    "nu-tre": "vi-VN-HoaiMyNeural", "nu-truyen-cam": "vi-VN-HoaiMyNeural", "nu-tin-nhanh": "vi-VN-HoaiMyNeural",
    "nam": "vi-VN-NamMinhNeural", "male": "vi-VN-NamMinhNeural", "namminh": "vi-VN-NamMinhNeural",
    "nam-tram": "vi-VN-NamMinhNeural", "nam-ban-tin": "vi-VN-NamMinhNeural", "nam-nang-dong": "vi-VN-NamMinhNeural",
}


def clean(s):
    import html as html_lib
    return re.sub(r"\s+", " ", html_lib.unescape(s or "")).strip()


def _domain(url):
    try:return urlparse(url or '').netloc.replace('www.','')
    except Exception:return ''


def _strip_bullet_prefix(s):
    return clean(re.sub(r'^[\s•\-\*·▪▫●○\d\.\)\(]+', '', s or ''))


def _source_badge_url_first(post):
    d=_domain(post.get('url',''))
    if d:return d
    for s in post.get('sources') or []:
        d=_domain(s.get('url',''))
        if d:return d
    return 'VNEWS'


def _abs_url(src, base_url):
    if not src:return ''
    src=src.strip()
    if src.startswith('//'):return 'https:'+src
    if src.startswith('/'):
        try:
            p=urlparse(base_url);return f'{p.scheme}://{p.netloc}{src}'
        except Exception:return src
    return src


def _article_content_block(soup):
    for tag in soup.find_all(['script','style','nav','footer','aside','form','noscript','iframe']):tag.decompose()
    # Aggressively remove related/ad/recommend containers before image collection.
    bad_re=re.compile(r'(related|relate|recommend|suggest|sidebar|ads|advert|popular|more|xem-them|xemthem|tin-lien-quan|tinlienquan|doc-them|docthem|other-news|news-other|article-related|box-tin|box_related|story-related|recommend-news|same-category|cate-list|news-list|most-view|banner|qc|quang-cao|sponsor)',re.I)
    for el in list(soup.find_all(True)):
        cls=' '.join(el.get('class',[])); eid=el.get('id',''); role=el.get('role','')
        if bad_re.search(cls) or bad_re.search(eid) or bad_re.search(role):
            el.decompose()
    selectors=['article','main article','.article-content','.article__body','.article-body','.article-detail','.detail-content','.content-detail','.singular-content','.news-content','.post-content','.entry-content','.knc-content','.fck_detail','.cms-body','.story-body','[class*=article-content]','[class*=detail-content]','[class*=singular-content]']
    for sel in selectors:
        el=soup.select_one(sel)
        if el and (len(el.find_all('p'))>=2 or len(el.find_all(['figure','picture','img']))>=1):return el
    best=None;score=0
    for el in soup.find_all(['article','main','section','div']):
        ps=el.find_all('p');imgs=el.find_all('img');txt=' '.join(p.get_text(' ',strip=True) for p in ps)
        sc=len(ps)*120+len(imgs)*10+min(len(txt),4500)
        cls=' '.join(el.get('class',[])).lower()
        if any(k in cls for k in ['article','content','detail','post','entry','story']):sc+=800
        if sc>score:best=el;score=sc
    return best or soup


def _image_is_likely_article(im, src):
    low=(src or '').lower()
    if not src or src.startswith('data:') or 'base64' in low:return False
    if any(x in low for x in ['logo','icon','avatar','sprite','banner','ads','advert','tracking','pixel','social','share','author','thumb-related']):return False
    alt=(im.get('alt') or im.get('title') or '').lower()
    if any(x in alt for x in ['logo','avatar','quảng cáo','advertisement','banner']):return False
    try:
        w=int(re.sub(r'\D','',str(im.get('width') or '0')) or 0);h=int(re.sub(r'\D','',str(im.get('height') or '0')) or 0)
        if (w and w<220) or (h and h<140):return False
    except Exception:pass
    return True


def _article_only_images(url):
    """Collect images only inside main article content. If uncertain, return fewer/no images rather than related/ad images."""
    imgs=[]
    try:
        from bs4 import BeautifulSoup
        r=requests.get(url,headers=getattr(base,'HEADERS',{}),timeout=18);r.encoding='utf-8'
        soup=BeautifulSoup(r.text,'lxml')
        block=_article_content_block(soup)
        candidates=[]
        # Prefer figure/picture under article body; then direct img in body.
        for el in block.find_all(['figure','picture'],recursive=True):
            im=el.find('img')
            if im:candidates.append(im)
        for im in block.find_all('img',recursive=True):
            if im not in candidates:candidates.append(im)
        seen=set()
        for im in candidates:
            src=(im.get('data-src') or im.get('data-original') or im.get('data-lazy-src') or im.get('data-srcset') or im.get('srcset') or im.get('src') or '')
            if ',' in src:src=src.split(',')[0].strip().split(' ')[0]
            else:src=src.strip().split(' ')[0]
            src=_abs_url(src,url)
            if src in seen or not _image_is_likely_article(im,src):continue
            # parent text guard: skip images from any remaining related block
            parent_txt=' '.join((im.parent.get('class',[]) if im.parent else []))+' '+(im.parent.get('id','') if im.parent else '')
            if re.search(r'(related|recommend|tin-lien-quan|doc-them|xem-them|popular|ads|banner)',parent_txt,re.I):continue
            seen.add(src);imgs.append(src)
            if len(imgs)>=20:break
        # Use og:image ONLY as article main image fallback when no body image found.
        if not imgs:
            og=soup.find('meta',property='og:image') or soup.find('meta',attrs={'name':'twitter:image'})
            if og:
                src=_abs_url(og.get('content',''),url)
                if src and 'logo' not in src.lower() and 'banner' not in src.lower():imgs.append(src)
    except Exception:pass
    return imgs[:20]


def _scrape_url_article_only(url):
    data=base.scrape_any_url(url)
    imgs=_article_only_images(url)
    data['images']=imgs
    if imgs:data['image']=imgs[0]
    else:data['image']=''
    return data


def _blank_image(path, title='VNEWS'):
    if Image is None:return None
    im=Image.new('RGB',(1080,760),(24,48,36));draw=ImageDraw.Draw(im)
    try:f=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',48)
    except Exception:f=None
    draw.text((60,330),clean(title)[:40] or 'VNEWS',fill=(255,255,255),font=f)
    im.save(path,quality=90);return path


def _download_image_safe(url, fallback_title, out_path):
    if url:
        try:
            r=requests.get(url,headers=getattr(base,'HEADERS',{}),timeout=18)
            if r.status_code==200 and len(r.content)>1200:
                with open(out_path,'wb') as f:f.write(r.content)
                # verify PIL opens it
                if Image:
                    Image.open(out_path).verify()
                return out_path
        except Exception:pass
    try:
        return base._download_image('',fallback_title,out_path)
    except Exception:
        return _blank_image(out_path,fallback_title)


def final_make_tts(text,voice,out_path):
    text=_strip_bullet_prefix(text) or 'Bản tin VNEWS.'
    # Only Vietnamese voices. Unknown choices fall back to Vietnamese female.
    edge_voice=VN_VOICES.get(str(voice or '').lower().strip(), 'vi-VN-HoaiMyNeural')
    for ev in [edge_voice, 'vi-VN-HoaiMyNeural', 'vi-VN-NamMinhNeural']:
        try:
            subprocess.run(['python','-m','edge_tts','--voice',ev,'--text',text,'--write-media',out_path],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=180)
            if os.path.exists(out_path) and os.path.getsize(out_path)>1000:return out_path
        except Exception:pass
    try:
        base.gTTS(text,lang='vi',tld='com.vn',slow=False).save(out_path)
        if os.path.exists(out_path) and os.path.getsize(out_path)>1000:return out_path
    except Exception:pass
    # Last-resort silent audio guarantees short generation succeeds.
    subprocess.run(['ffmpeg','-y','-f','lavfi','-i','anullsrc=channel_layout=stereo:sample_rate=44100','-t','3','-q:a','9','-acodec','libmp3lame',out_path],stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=30)
    return out_path


def _draw_center(draw, lines, font, y, fill, W, line_h):
    for ln in lines:
        try:box=draw.textbbox((0,0),ln,font=font);tw=box[2]-box[0]
        except Exception:tw=len(ln)*24
        draw.text((max(30,(W-tw)//2),y),ln,fill=fill,font=font);y+=line_h
    return y


def final_make_frame(post,seg,idx,total,img_path,out_path):
    if Image is None:return rt.make_frame(post,seg,idx,total,img_path,out_path)
    W,H=1080,1920;hero_h=760;bg=Image.new('RGB',(W,H),(12,12,12))
    try:
        im=Image.open(img_path).convert('RGB');ratio=im.width/max(1,im.height);tr=W/hero_h
        if ratio>tr:nh=hero_h;nw=int(nh*ratio)
        else:nw=W;nh=int(nw/ratio)
        im=im.resize((nw,nh));left=(nw-W)//2;top=(nh-hero_h)//2;bg.paste(im.crop((left,top,left+W,top+hero_h)),(0,0))
    except Exception:pass
    draw=ImageDraw.Draw(bg)
    try:
        fb=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',58);ft=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',38);fs=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf',30);fsmall=ImageFont.truetype('/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf',28)
    except Exception:fb=ft=fs=fsmall=None
    badge='Nguồn: '+_source_badge_url_first(post)
    try:b=draw.textbbox((0,0),badge,font=fsmall);bw=b[2]-b[0];bh=b[3]-b[1]
    except Exception:bw=len(badge)*16;bh=34
    bx=W-bw-42;by=24;draw.rounded_rectangle((bx-16,by-8,W-24,by+bh+14),radius=18,fill=(0,0,0));draw.text((bx,by),badge,fill=(255,255,255),font=fsmall)
    draw.rectangle((0,hero_h-20,W,H),fill=(12,12,12))
    total=max(1,total);total_w=total*38-14;start=(W-total_w)//2
    for i in range(total):draw.rounded_rectangle((start+i*38,820,start+i*38+24,832),radius=6,fill=(92,184,122) if i==idx else (70,70,70))
    brand='VNEWS AI SHORT'
    try:bb=draw.textbbox((0,0),brand,font=ft);tx=(W-(bb[2]-bb[0]))//2
    except Exception:tx=360
    draw.text((tx,870),brand,fill=(110,231,143),font=ft)
    seg=_strip_bullet_prefix(seg);lines=rt.wrap_text(draw,seg,fb,W-120,8);y=max(980,1250-(len(lines)*74)//2);_draw_center(draw,lines,fb,y,(255,255,255),W,74)
    title_lines=rt.wrap_text(draw,_strip_bullet_prefix(post.get('title','')),fs,W-120,3);y2=1640;draw.line((80,y2-26,W-80,y2-26),fill=(70,70,70),width=2);_draw_center(draw,title_lines,fs,y2,(220,220,220),W,42)
    bg.save(out_path,quality=92)

# Monkey patches for old functions.
rt.make_frame=final_make_frame;rt.make_tts=final_make_tts;rt._source_badge=_source_badge_url_first

# Override endpoints.
_PATCH={('/api/url_wall','POST'),('/api/rewrite_share','POST'),('/api/ai/url','POST'),('/api/ai/short/{post_id}','POST'),('/','GET'),('/aw','GET')}
app.router.routes=[r for r in app.router.routes if not any(getattr(r,'path',None)==p and m in getattr(r,'methods',set()) for p,m in _PATCH)]

@app.post('/api/url_wall')
async def final_url_wall(request:Request):
    body=await request.json();url=base._clean_text(body.get('url',''))
    if not url.startswith('http'):return JSONResponse({'error':'missing url'},status_code=400)
    try:data=_scrape_url_article_only(url)
    except Exception as e:return JSONResponse({'error':'Không scrape được URL: '+str(e)[:180]},status_code=422)
    raw=(data.get('summary','')+'\n'+data.get('text','')).strip()
    if len(raw)<120:return JSONResponse({'error':'URL không có đủ nội dung để tóm tắt'},status_code=422)
    prompt=f"""Tóm tắt bài viết nguồn dưới đây để đăng lên Tường AI VNEWS.

Yêu cầu:
- Chỉ tóm tắt nội dung chính, không viết lại toàn bộ bài.
- Ngắn gọn, cụ thể, dễ hiểu.
- Không lặp ý, không thêm chi tiết ngoài nguồn.
- Tối đa 5 ý chính hoặc 2 đoạn ngắn.
- Hạn chế dùng dấu đầu dòng.

Tiêu đề gốc: {data.get('title','')}
Nguồn: {_domain(url)}
Nội dung gốc:
{raw[:16000]}"""
    text=await base.qwen_generate(prompt,image_url=(data.get('image') or None),max_tokens=900)
    if not text:text=rt.old._fallback_summary_from_prompt(prompt,max_units=5) if hasattr(rt.old,'_fallback_summary_from_prompt') else raw[:900]
    text=rt.postprocess(text) if hasattr(rt,'postprocess') else text
    src=[{'title':data.get('title'), 'url':url, 'excerpt':raw[:500], 'via':_domain(url)}]
    if 'Nguồn tham khảo:' not in text:text+='\n\n'+rt.source_line(src)
    imgs=data.get('images') or []
    post=base.make_post(data.get('title') or 'Bài viết',text,imgs[0] if imgs else '',url,'url',sources=src)
    post['images']=imgs
    posts=base._load_ai_wall();posts.insert(0,post);base._save_ai_wall(posts)
    return JSONResponse({'post':post})

@app.post('/api/rewrite_share')
async def final_rewrite_share(request:Request):return await final_url_wall(request)
@app.post('/api/ai/url')
async def final_ai_url(request:Request):return await final_url_wall(request)

@app.post('/api/ai/short/{post_id}')
async def final_short(post_id:str,request:Request):
    try:body=await request.json()
    except Exception:body={}
    voice=str(body.get('voice','nu')).lower().strip();emotion=str(body.get('emotion','neutral')).lower().strip();speed=max(0.85,min(1.35,float(body.get('speed',1.2) or 1.2)))
    posts=base._load_ai_wall();post=next((p for p in posts if str(p.get('id'))==str(post_id)),None)
    if not post:return JSONResponse({'error':'post not found'},status_code=404)
    segs=rt.split_segments(post,8) if hasattr(rt,'split_segments') else [_strip_bullet_prefix(post.get('text') or post.get('title') or 'VNEWS')]
    imgs=[u for u in (post.get('images') or []) if u] or ([post.get('img')] if post.get('img') else [])
    os.makedirs(base.SHORTS_DIR,exist_ok=True);suffix=f'_{voice}_{emotion}_{str(speed).replace(".","p")}_articleimgs_vivoice'
    out=os.path.join(base.SHORTS_DIR,base._safe_name(post_id+suffix)+'.mp4')
    if os.path.exists(out):
        post['video']='/api/ai/short-file/'+post_id+suffix;base._save_ai_wall(posts);return JSONResponse({'video':post['video'],'segments':len(segs),'subtitles':False})
    work=os.path.join(base.SHORTS_DIR,base._safe_name(post_id+suffix));os.makedirs(work,exist_ok=True)
    clips=[]
    try:
        for i,seg in enumerate(segs):
            img_url=imgs[i % len(imgs)] if imgs else ''
            img=os.path.join(work,f'image_{i}.jpg');frame=os.path.join(work,f'f{i}.jpg');aud=os.path.join(work,f'a{i}.mp3');aud2=os.path.join(work,f'a{i}_fast.mp3');clip=os.path.join(work,f'c{i}.mp4')
            _download_image_safe(img_url,post.get('title','AI news'),img)
            seg=_strip_bullet_prefix(seg);final_make_frame(post,seg,i,len(segs),img,frame)
            prefix={'urgent':'Tin nhanh.','warm':'Câu chuyện đáng chú ý.','serious':'Bản tin nghiêm túc.','energetic':'Cập nhật nổi bật.'}.get(emotion,'')
            spoken=(prefix+' '+seg).strip() if i==0 and prefix else seg
            final_make_tts(spoken,voice,aud)
            try:subprocess.run(['ffmpeg','-y','-i',aud,'-filter:a',f'atempo={speed}','-vn',aud2],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=120)
            except Exception:aud2=aud
            try:
                subprocess.run(['ffmpeg','-y','-loop','1','-i',frame,'-i',aud2,'-shortest','-c:v','libx264','-tune','stillimage','-pix_fmt','yuv420p','-c:a','aac','-b:a','128k','-vf','scale=1080:1920',clip],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=180)
            except Exception:
                # last-resort visual-only 4s clip
                subprocess.run(['ffmpeg','-y','-loop','1','-t','4','-i',frame,'-f','lavfi','-i','anullsrc=channel_layout=stereo:sample_rate=44100','-shortest','-c:v','libx264','-pix_fmt','yuv420p','-c:a','aac','-vf','scale=1080:1920',clip],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=120)
            clips.append(clip)
        lf=os.path.join(work,'list.txt')
        with open(lf,'w',encoding='utf-8') as f:
            for c in clips:f.write("file '"+c.replace("","'\\''"))+"'\n")
        subprocess.run(['ffmpeg','-y','-f','concat','-safe','0','-i',lf,'-c','copy',out],check=True,stdout=subprocess.PIPE,stderr=subprocess.PIPE,timeout=240)
        post['video']='/api/ai/short-file/'+post_id+suffix;post['short_subtitles']=False;post['short_segments']=segs;post['short_speed']=speed;base._save_ai_wall(posts)
        return JSONResponse({'video':post['video'],'segments':len(segs),'subtitles':False})
    except Exception as e:return JSONResponse({'error':'Không tạo được shorts: '+str(e)[:220]},status_code=500)

@app.get('/aw')
def ai_wall_share(post:str=Query(default=''), short:int=Query(default=0)):
    posts=base._load_ai_wall();p=next((x for x in posts if str(x.get('id'))==str(post)),None)
    if not p:return HTMLResponse(f'<script>location.href="{SPACE_URL}"</script>')
    title=p.get('title') or 'VNEWS AI';img=p.get('img') or DEFAULT_IMG
    desc=(p.get('text') or '')[:220]
    return HTMLResponse(f'<!doctype html><html><head><meta charset="utf-8"><title>{title}</title><meta property="og:title" content="{title}"><meta property="og:description" content="{desc}"><meta property="og:image" content="{img}"><meta property="og:type" content="article"><meta name="twitter:card" content="summary_large_image"></head><body><script>localStorage.setItem('pending_ai_post','{post}');location.href='{SPACE_URL}'</script></body></html>')

FINAL_INJECT = r'''
<style>