HoaHuggingFace commited on
Commit
b0db312
·
verified ·
1 Parent(s): d9b7963

Upload 2 files

Browse files
Files changed (2) hide show
  1. app.py +1165 -0
  2. requirements.txt +6 -0
app.py ADDED
@@ -0,0 +1,1165 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import importlib
2
+ import subprocess
3
+ import sys
4
+ import zipfile
5
+ import os
6
+ import time
7
+ import torch
8
+ import gradio as gr
9
+ from moviepy.editor import VideoFileClip, AudioFileClip
10
+ from pydub import AudioSegment
11
+ import shutil
12
+ import atexit
13
+
14
+ def install_package(name, pkg_type, upgrade=False, import_name=None, check_cmd=None):
15
+ if pkg_type == "pip":
16
+ import_name = import_name or name.replace("-", "_")
17
+ try:
18
+ importlib.import_module(import_name)
19
+ if upgrade:
20
+ print(f"⏫ Nâng cấp {name} ...")
21
+ subprocess.check_call([sys.executable, "-m", "pip", "install", "--upgrade", name])
22
+ else:
23
+ print(f"✅ Đã có sẵn: {name}")
24
+ except ImportError:
25
+ print(f"⏳ Đang cài đặt: {name} ...")
26
+ subprocess.check_call([sys.executable, "-m", "pip", "install", name])
27
+
28
+ # Chỉ sửa từ phía dưới này.
29
+ install_package("gradio", "pip", upgrade=False)
30
+ install_package("faster-whisper", "pip", upgrade=False)
31
+ install_package("yt-dlp", "pip", upgrade=False, import_name="yt_dlp")
32
+
33
+ import os
34
+ import time
35
+ import torch
36
+ import gradio as gr
37
+ from moviepy.editor import VideoFileClip, AudioFileClip
38
+ from pydub import AudioSegment
39
+ import shutil
40
+ import atexit
41
+
42
+ # ====== LAZY LOADING: Không load model ngay ======
43
+ model = None
44
+ batched_model = None
45
+ device = "cuda" if torch.cuda.is_available() else "cpu"
46
+ compute_type = "float16" if torch.cuda.is_available() else "float32"
47
+ model_size = "large-v3"
48
+
49
+ # Hallucination blocklist
50
+ HALLUCINATION_BLOCKLIST = [
51
+ "Hãy subscribe cho kênh Ghiền Mì Gõ",
52
+ "Để không bỏ lỡ những video hấp dẫn",
53
+ "Hãy đăng ký kênh để ủng hộ kênh của mình nhé",
54
+ "Cảm ơn các bạn đã theo dõi",
55
+ ]
56
+
57
+ # ====== [WEBM] Các định dạng được coi là "audio-only" (xử lý như audio, bỏ qua video track) ======
58
+ WEBM_AS_AUDIO_EXTS = {'.webm'}
59
+
60
+ def load_whisper_model():
61
+ """Load model chỉ khi cần thiết"""
62
+ global model, batched_model
63
+ if model is None:
64
+ print("🔄 Đang load Whisper model...")
65
+ from faster_whisper import WhisperModel, BatchedInferencePipeline
66
+ model = WhisperModel(
67
+ model_size,
68
+ device=device,
69
+ compute_type=compute_type,
70
+ )
71
+ batched_model = BatchedInferencePipeline(model=model)
72
+ print("✅ Model đã load xong!")
73
+ return model, batched_model
74
+
75
+ # Lưu danh sách temp directories để cleanup sau
76
+ temp_dirs = []
77
+
78
+ def cleanup_temp_files():
79
+ """Xóa tất cả temporary files khi thoát"""
80
+ for temp_dir in temp_dirs:
81
+ try:
82
+ if os.path.exists(temp_dir):
83
+ shutil.rmtree(temp_dir)
84
+ except Exception as e:
85
+ print(f"Không thể xóa {temp_dir}: {e}")
86
+
87
+ # Đăng ký cleanup khi thoát
88
+ atexit.register(cleanup_temp_files)
89
+
90
+ def format_timestamp(seconds, include_milliseconds=True):
91
+ """Format timestamp từ giây sang HH:MM:SS.mmm hoặc HH:MM:SS"""
92
+ if seconds is None:
93
+ return "00:00:00.000"
94
+ h, m, s = int(seconds) // 3600, (int(seconds) % 3600) // 60, int(seconds) % 60
95
+ if include_milliseconds:
96
+ ms = int((seconds - int(seconds)) * 1000)
97
+ return f"{h:02d}:{m:02d}:{s:02d}.{ms:03d}"
98
+ else:
99
+ return f"{h:02d}:{m:02d}:{s:02d}"
100
+
101
+ def webm_to_mp3(webm_path, output_dir=None):
102
+ """
103
+ [WEBM] Convert file .webm sang .mp3 bằng FFmpeg (extract audio track).
104
+ Trả về đường dẫn file mp3 đã tạo.
105
+ """
106
+ if output_dir is None:
107
+ output_dir = os.path.dirname(webm_path)
108
+ os.makedirs(output_dir, exist_ok=True)
109
+ file_basename = os.path.splitext(os.path.basename(webm_path))[0]
110
+ mp3_path = os.path.join(output_dir, f"{file_basename}.mp3")
111
+ cmd = [
112
+ 'ffmpeg',
113
+ '-i', webm_path,
114
+ '-vn', # Bỏ video track
115
+ '-acodec', 'libmp3lame',
116
+ '-q:a', '2',
117
+ mp3_path,
118
+ '-y'
119
+ ]
120
+ result = subprocess.run(cmd, capture_output=True)
121
+ if result.returncode != 0:
122
+ raise RuntimeError(
123
+ f"FFmpeg không thể convert WebM sang MP3:\n{result.stderr.decode(errors='replace')}"
124
+ )
125
+ return mp3_path
126
+
127
+ def get_duration(file_path):
128
+ """Lấy duration từ file"""
129
+ if not file_path:
130
+ return 100
131
+
132
+ clip = None
133
+ try:
134
+ file_ext = os.path.splitext(file_path)[1].lower()
135
+
136
+ # [WEBM] Dùng FFmpeg probe thay vì MoviePy để tránh lỗi codec
137
+ if file_ext in WEBM_AS_AUDIO_EXTS:
138
+ result = subprocess.run(
139
+ ['ffprobe', '-v', 'error', '-show_entries', 'format=duration',
140
+ '-of', 'default=noprint_wrappers=1:nokey=1', file_path],
141
+ capture_output=True, text=True
142
+ )
143
+ duration_str = result.stdout.strip()
144
+ return float(duration_str) if duration_str else 100
145
+
146
+ if file_ext in ['.mp4', '.mkv', '.avi', '.mov', '.flv']:
147
+ clip = VideoFileClip(file_path)
148
+ duration = clip.duration
149
+ clip.close()
150
+ else:
151
+ audio = AudioSegment.from_file(file_path)
152
+ duration = len(audio) / 1000.0
153
+ return duration
154
+ except:
155
+ return 100
156
+ finally:
157
+ if clip:
158
+ clip.close()
159
+
160
+ def download_and_convert_to_mp3(url):
161
+ """Tải video từ URL và convert sang MP3 (giữ lại video gốc)"""
162
+ if not url or not url.strip():
163
+ return None, None, "", 999999, 0
164
+
165
+ clip = None
166
+ try:
167
+ import yt_dlp
168
+ temp_dir = os.path.join(os.getcwd(), "temp_downloads")
169
+ os.makedirs(temp_dir, exist_ok=True)
170
+ temp_dirs.append(temp_dir)
171
+
172
+ # Tải video/audio tốt nhất
173
+ ydl_opts = {
174
+ 'format': 'bestvideo+bestaudio/best',
175
+ 'outtmpl': os.path.join(temp_dir, '%(title)s.%(ext)s'),
176
+ 'quiet': True,
177
+ 'no_warnings': True,
178
+ 'http_headers': {
179
+ 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'
180
+ }
181
+ }
182
+
183
+ yield None, None, "Đang tải video...", 999999, 0
184
+
185
+ with yt_dlp.YoutubeDL(ydl_opts) as ydl:
186
+ info = ydl.extract_info(url)
187
+ video_filename = ydl.prepare_filename(info)
188
+ video_title = info.get('title', 'audio')
189
+ uploader = info.get('uploader', 'N/A')
190
+ view_count = info.get('view_count', 0)
191
+
192
+ yield None, None, "Đang chuyển đổi sang MP3...", 999999, 0
193
+
194
+ # Convert sang MP3
195
+ file_ext = os.path.splitext(video_filename)[1].lower()
196
+ mp3_filename = os.path.join(temp_dir, f"{video_title}.mp3")
197
+
198
+ # [WEBM] Dùng FFmpeg extract audio, không dùng MoviePy
199
+ if file_ext in WEBM_AS_AUDIO_EXTS:
200
+ mp3_filename = webm_to_mp3(video_filename, output_dir=temp_dir)
201
+ duration = get_duration(video_filename)
202
+ elif file_ext in ['.mp4', '.mkv', '.avi', '.mov', '.flv']:
203
+ clip = VideoFileClip(video_filename)
204
+ clip.audio.write_audiofile(mp3_filename, logger=None)
205
+ duration = clip.duration
206
+ clip.close()
207
+ else:
208
+ audio = AudioSegment.from_file(video_filename)
209
+ audio.export(mp3_filename, format='mp3')
210
+ duration = len(audio) / 1000.0
211
+
212
+ info_text = f"""📹 **Tiêu đề:** {video_title}
213
+ 👤 **Kênh:** {uploader}
214
+ 👁️ **Lượt xem:** {view_count:,}
215
+ ⏱️ **Thời lượng:** {int(duration//60)}:{int(duration%60):02d} ({duration:.1f} giây)
216
+ ✅ **Đã convert sang MP3**"""
217
+
218
+ # Trả về cả MP3 (để transcribe) và video gốc (để trim)
219
+ yield (
220
+ mp3_filename,
221
+ video_filename,
222
+ info_text,
223
+ duration,
224
+ 0
225
+ )
226
+
227
+ except Exception as e:
228
+ yield None, None, f"❌ Lỗi: {str(e)}", 999999, 0
229
+
230
+ finally:
231
+ if clip:
232
+ clip.close()
233
+
234
+ def process_upload(file_path):
235
+ """Xử lý file upload"""
236
+ if not file_path:
237
+ return None, None, "", 999999, 0
238
+
239
+ clip = None
240
+ try:
241
+ file_ext = os.path.splitext(file_path)[1].lower()
242
+ file_name = os.path.basename(file_path)
243
+ file_size = os.path.getsize(file_path) / (1024 * 1024) # MB
244
+
245
+ yield None, None, "Đang xử lý file upload...", 999999, 0
246
+
247
+ # [WEBM] Xử lý như audio: extract MP3 bằng FFmpeg, không dùng VideoFileClip
248
+ if file_ext in WEBM_AS_AUDIO_EXTS:
249
+ temp_dir = os.path.join(os.getcwd(), "temp_webm_upload")
250
+ os.makedirs(temp_dir, exist_ok=True)
251
+ temp_dirs.append(temp_dir)
252
+
253
+ mp3_path = webm_to_mp3(file_path, output_dir=temp_dir)
254
+ duration = get_duration(file_path)
255
+ file_type = "Audio (WebM)"
256
+ video_file = None # Không treat như video, không hiện preview video
257
+
258
+ info_text = f"""📁 **Tên file:** {file_name}
259
+ 🎵 **Loại:** {file_type}
260
+ 💾 **Kích thước:** {file_size:.2f} MB
261
+ ⏱️ **Thời lượng:** {int(duration//60)}:{int(duration%60):02d} ({duration:.1f}s)
262
+ ✅ **Đã extract audio sang MP3**"""
263
+
264
+ yield (
265
+ mp3_path, # current_file → MP3 để transcribe & trim
266
+ video_file, # video_file_state → None (không có video track)
267
+ info_text,
268
+ duration,
269
+ 0
270
+ )
271
+ return
272
+
273
+ if file_ext in ['.mp4', '.mkv', '.avi', '.mov', '.flv']:
274
+ clip = VideoFileClip(file_path)
275
+ duration = clip.duration
276
+ clip.close()
277
+ file_type = "Video"
278
+ video_file = file_path
279
+ elif file_ext in ['.mp3', '.wav', '.m4a', '.ogg', '.flac', '.aac']:
280
+ audio = AudioSegment.from_file(file_path)
281
+ duration = len(audio) / 1000.0
282
+ file_type = "Audio"
283
+ video_file = None
284
+ else:
285
+ yield None, None, f"⚠️ Định dạng file không đ��ợc hỗ trợ: {file_ext}", 999999, 0
286
+ return
287
+
288
+ info_text = f"""📁 **Tên file:** {file_name}
289
+ 🎬 **Loại:** {file_type}
290
+ 💾 **Kích thước:** {file_size:.2f} MB
291
+ ⏱️ **Thời lượng:** {int(duration//60)}:{int(duration%60):02d} ({duration:.1f}s)"""
292
+
293
+ yield (
294
+ file_path,
295
+ video_file,
296
+ info_text,
297
+ duration,
298
+ 0
299
+ )
300
+
301
+ except Exception as e:
302
+ yield None, None, f"❌ Lỗi: {str(e)}", 999999, 0
303
+ finally:
304
+ if clip:
305
+ clip.close()
306
+
307
+ def convert_to_mp3(audio_file, video_file, bitrate, sample_rate):
308
+ """
309
+ Convert audio/video sang MP3 với bitrate và sample rate tùy chọn (tối ưu cho transcribe).
310
+ """
311
+ if not audio_file and not video_file:
312
+ yield None, "⚠️ Không có file để convert."
313
+ return
314
+
315
+ try:
316
+ input_file = video_file if video_file else audio_file
317
+ file_basename = os.path.splitext(os.path.basename(input_file))[0]
318
+
319
+ temp_dir = os.path.join(os.getcwd(), "temp_converted")
320
+ os.makedirs(temp_dir, exist_ok=True)
321
+ temp_dirs.append(temp_dir)
322
+
323
+ mp3_file = os.path.join(temp_dir, f"{file_basename}_{bitrate}_{sample_rate}hz.mp3")
324
+
325
+ yield None, "⏳ Đang convert sang MP3..."
326
+
327
+ cmd = [
328
+ 'ffmpeg',
329
+ '-i', input_file,
330
+ '-vn',
331
+ '-acodec', 'libmp3lame',
332
+ '-b:a', bitrate,
333
+ '-ar', str(sample_rate),
334
+ mp3_file,
335
+ '-y'
336
+ ]
337
+ result = subprocess.run(cmd, capture_output=True)
338
+ if result.returncode != 0:
339
+ yield None, f"❌ Lỗi FFmpeg: {result.stderr.decode(errors='replace')}"
340
+ return
341
+
342
+ file_size_kb = os.path.getsize(mp3_file) / 1024
343
+ status_msg = (
344
+ f"✅ Convert thành công!\n"
345
+ f"📁 File: {os.path.basename(mp3_file)}\n"
346
+ f"🎵 Bitrate: {bitrate} | Sample rate: {sample_rate} Hz\n"
347
+ f"💾 Kích thước: {file_size_kb:.1f} KB"
348
+ )
349
+ yield mp3_file, status_msg
350
+
351
+ except Exception as e:
352
+ import traceback
353
+ yield None, f"❌ Lỗi: {str(e)}\n{traceback.format_exc()}"
354
+
355
+ def generate_srt(segments, start_offset=0):
356
+ """Tạo nội dung file SRT từ segments"""
357
+ srt_content = []
358
+ for i, segment in enumerate(segments, start=1):
359
+ start_time = segment.start + start_offset
360
+ end_time = segment.end + start_offset
361
+ start_ts = format_timestamp(start_time).replace('.', ',')
362
+ end_ts = format_timestamp(end_time).replace('.', ',')
363
+ srt_content.append(f"{i}")
364
+ srt_content.append(f"{start_ts} --> {end_ts}")
365
+ srt_content.append(segment.text.strip())
366
+ srt_content.append("")
367
+ return "\n".join(srt_content)
368
+
369
+ def generate_vtt(segments, start_offset=0):
370
+ """Tạo nội dung file VTT từ segments"""
371
+ vtt_content = ["WEBVTT", ""]
372
+ for segment in segments:
373
+ start_time = segment.start + start_offset
374
+ end_time = segment.end + start_offset
375
+ start_ts = format_timestamp(start_time)
376
+ end_ts = format_timestamp(end_time)
377
+ vtt_content.append(f"{start_ts} --> {end_ts}")
378
+ vtt_content.append(segment.text.strip())
379
+ vtt_content.append("")
380
+ return "\n".join(vtt_content)
381
+
382
+ def format_transcript_display(raw_segments, include_timestamps_value, original_trim_start_time):
383
+ """Formats the transcript text based on stored segments and timestamp preference."""
384
+ if not raw_segments:
385
+ return ""
386
+
387
+ current_transcript_lines = []
388
+ for segment in raw_segments:
389
+ actual_start = segment.start + original_trim_start_time
390
+ actual_end = segment.end + original_trim_start_time
391
+
392
+ text = segment.text.strip()
393
+ for block_phrase in HALLUCINATION_BLOCKLIST:
394
+ if block_phrase in text:
395
+ text = text.replace(block_phrase, "").strip()
396
+ print(f"Removed hallucination: '{block_phrase}' from segment.")
397
+
398
+ if include_timestamps_value:
399
+ start_ts = format_timestamp(actual_start, include_milliseconds=False)
400
+ end_ts = format_timestamp(actual_end, include_milliseconds=False)
401
+ current_transcript_lines.append(f"[{start_ts} → {end_ts}] {text}")
402
+ else:
403
+ current_transcript_lines.append(text)
404
+ return "\n".join(current_transcript_lines)
405
+
406
+ def transcribe_audio(input_file, video_file, method, beam_size, start_time, end_time, include_timestamps, batch_size, min_silence_duration_ms, speech_pad_ms, no_speech_threshold, condition_on_previous_text, minimum_speech_duration):
407
+ """Transcribe audio file với các tùy chọn nâng cao (đã tối ưu)"""
408
+ if not input_file:
409
+ yield ("⚠️ Không có file audio. Vui lòng upload hoặc nhập URL YouTube.", None, None, None, None, None, None, None, 0)
410
+ return
411
+
412
+ try:
413
+ # Load model khi cần
414
+ model, batched_model = load_whisper_model()
415
+
416
+ beam_size = int(beam_size)
417
+ batch_size = int(batch_size)
418
+ min_silence_duration_ms = int(min_silence_duration_ms)
419
+ speech_pad_ms = int(speech_pad_ms)
420
+ no_speech_threshold = float(no_speech_threshold)
421
+ minimum_speech_duration = float(minimum_speech_duration)
422
+
423
+ start_transcribe = time.time()
424
+
425
+ # Xác định file cần transcribe (trimmed hoặc full)
426
+ duration = get_duration(input_file)
427
+ is_trimmed = not (start_time == 0 and end_time >= duration)
428
+
429
+ # Nếu có trim, tạo file trimmed trước khi transcribe
430
+ actual_start_offset_for_transcription = 0
431
+ if is_trimmed:
432
+ if start_time >= end_time:
433
+ yield ("❌ Thời gian bắt đầu phải nhỏ hơn thời gian kết thúc.", None, None, None, None, None, None, None, 0)
434
+ return
435
+
436
+ temp_dir_trim_for_transcribe = os.path.join(os.getcwd(), "temp_transcribe_for_stream")
437
+ os.makedirs(temp_dir_trim_for_transcribe, exist_ok=True)
438
+ temp_dirs.append(temp_dir_trim_for_transcribe)
439
+
440
+ file_basename = os.path.splitext(os.path.basename(input_file))[0]
441
+ import datetime
442
+ ts = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")
443
+ trimmed_file = os.path.join(temp_dir_trim_for_transcribe, f"{file_basename}_transcribe_{ts}.mp3")
444
+
445
+ # Trim bằng FFmpeg (nhanh nhất) — input_file lúc này luôn là .mp3 (kể cả từ WebM)
446
+ cmd = [
447
+ 'ffmpeg',
448
+ '-ss', str(start_time),
449
+ '-i', input_file,
450
+ '-t', str(end_time - start_time),
451
+ '-acodec', 'libmp3lame',
452
+ '-q:a', '2',
453
+ trimmed_file,
454
+ '-y'
455
+ ]
456
+ subprocess.run(cmd, capture_output=True, check=True)
457
+ audio_to_transcribe = trimmed_file
458
+ actual_start_offset_for_transcription = start_time
459
+ else:
460
+ audio_to_transcribe = input_file
461
+
462
+ # Transcribe với VAD optimization
463
+ segments_generator = None
464
+ info = None
465
+ if method == "model.transcribe":
466
+ segments_generator, info = model.transcribe(
467
+ audio_to_transcribe,
468
+ beam_size=beam_size,
469
+ vad_filter=True,
470
+ vad_parameters=dict(
471
+ min_silence_duration_ms=min_silence_duration_ms,
472
+ speech_pad_ms=speech_pad_ms
473
+ ),
474
+ temperature=0,
475
+ word_timestamps=True,
476
+ condition_on_previous_text=condition_on_previous_text
477
+ )
478
+ else:
479
+ segments_generator, info = batched_model.transcribe(
480
+ audio_to_transcribe,
481
+ beam_size=beam_size,
482
+ batch_size=batch_size,
483
+ temperature=0,
484
+ condition_on_previous_text=condition_on_previous_text
485
+ )
486
+
487
+ # --- Streaming part ---
488
+ current_transcript_lines = []
489
+ full_text_list = []
490
+ all_segments_for_files = []
491
+
492
+ yield ("Bắt đầu transcription...", None, None, None, None, None, None, None, 0)
493
+
494
+ for i, segment in enumerate(segments_generator):
495
+ all_segments_for_files.append(segment)
496
+
497
+ actual_start_stream = segment.start + actual_start_offset_for_transcription
498
+ actual_end_stream = segment.end + actual_start_offset_for_transcription
499
+
500
+ text = segment.text.strip()
501
+ for block_phrase in HALLUCINATION_BLOCKLIST:
502
+ if block_phrase in text:
503
+ text = text.replace(block_phrase, "").strip()
504
+ print(f"Removed hallucination: '{block_phrase}' from segment.")
505
+
506
+ if include_timestamps:
507
+ start_ts = format_timestamp(actual_start_stream, include_milliseconds=False)
508
+ end_ts = format_timestamp(actual_end_stream, include_milliseconds=False)
509
+ current_transcript_lines.append(f"[{start_ts} → {end_ts}] {text}")
510
+ else:
511
+ current_transcript_lines.append(text)
512
+
513
+ full_text_list.append(text)
514
+
515
+ yield ("\n".join(current_transcript_lines), None, None, None, None, None, None, None, 0)
516
+
517
+ # --- End of Streaming part, now process final files ---
518
+
519
+ if not all_segments_for_files:
520
+ yield ("⚠️ Không có nội dung được transcribe. Vui lòng kiểm tra file hoặc tùy chọn cắt.", None, None, None, None, None, None, None, 0)
521
+ return
522
+
523
+ transcript_text = format_transcript_display(all_segments_for_files, include_timestamps, actual_start_offset_for_transcription)
524
+
525
+ srt_content = generate_srt(all_segments_for_files, start_offset=(actual_start_offset_for_transcription))
526
+ vtt_content = generate_vtt(all_segments_for_files, start_offset=(actual_start_offset_for_transcription))
527
+
528
+ transcribe_time = time.time() - start_transcribe
529
+ detected_lang = info.language if hasattr(info, 'language') else "Unknown"
530
+ lang_prob = info.language_probability if hasattr(info, 'language_probability') else 0
531
+
532
+ file_basename = os.path.splitext(os.path.basename(input_file))[0]
533
+ import datetime
534
+ ts = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")
535
+ file_suffix = f"_{ts}"
536
+
537
+ # Save files
538
+ txt_file = f"{file_basename}{file_suffix}.txt"
539
+ with open(txt_file, "w", encoding="utf-8") as f:
540
+ f.write(f"=== TRANSCRIPTION INFO ===\n")
541
+ f.write(f"Model: {model_size}\n")
542
+ f.write(f"Compute type: {compute_type}\n")
543
+ f.write(f"Device: {device}\n")
544
+ f.write(f"Method: {method}\n")
545
+ f.write(f"Beam size: {beam_size}\n")
546
+ f.write(f"Batch size (BatchedInferencePipeline): {batch_size}\n")
547
+ f.write(f"VAD min_silence_duration_ms: {min_silence_duration_ms}\n")
548
+ f.write(f"VAD speech_pad_ms: {speech_pad_ms}\n")
549
+ f.write(f"VAD no_speech_threshold: {no_speech_threshold} (Not applicable for model.transcribe)\n")
550
+ f.write(f"VAD minimum_speech_duration: {minimum_speech_duration} (Not applicable for model.transcribe)\n")
551
+ f.write(f"Condition on previous text: {condition_on_previous_text}\n")
552
+ f.write(f"Language: {detected_lang} (confidence: {lang_prob:.2%})\n")
553
+ f.write(f"Processing time: {transcribe_time:.2f}s\n")
554
+ f.write(f"File: {file_basename}\n")
555
+ if is_trimmed:
556
+ f.write(f"Trimmed: {start_time}s - {end_time}s ({end_time - start_time:.1f}s)\n")
557
+ else:
558
+ f.write(f"Full file transcription\n")
559
+ f.write(f"\n=== TIMESTAMPED TRANSCRIPT ===\n")
560
+ f.write(transcript_text)
561
+ f.write(f"\n\n=== FULL TEXT ===\n")
562
+ f.write(" ".join(full_text_list))
563
+ f.write(f"\n\n=== METADATA ===\n")
564
+ f.write(f"Total segments: {len(all_segments_for_files)}\n")
565
+ f.write(f"Total characters: {len(' '.join(full_text_list))}\n")
566
+
567
+ srt_file = f"{file_basename}{file_suffix}.srt"
568
+ with open(srt_file, "w", encoding="utf-8") as f:
569
+ f.write(srt_content)
570
+
571
+ vtt_file = f"{file_basename}{file_suffix}.vtt"
572
+ with open(vtt_file, "w", encoding="utf-8") as f:
573
+ f.write(vtt_content)
574
+
575
+ info_display = f"""
576
+ ⏱️ Processing time : {transcribe_time:.2f}s
577
+ 🌐 Language : {detected_lang} ({lang_prob:.2%})
578
+ 📊 Segments : {len(all_segments_for_files)}
579
+ {'✂️ Trimmed :' + str(start_time) + 's - ' + str(end_time) + 's' if is_trimmed else '📄 Full file transcription'}
580
+ 💻 Device : {device}
581
+ 🧠 Model : {model_size}
582
+ ⚙️ Compute type : {compute_type}
583
+ Batch size (BatchedInferencePipeline) : {batch_size}
584
+ VAD min_silence_duration_ms : {min_silence_duration_ms}
585
+ VAD speech_pad_ms : {speech_pad_ms}
586
+ VAD no_speech_threshold : {no_speech_threshold} (Not applicable for model.transcribe)
587
+ VAD minimum_speech_duration : {minimum_speech_duration} (Not applicable for model.transcribe)
588
+ Condition on previous text : {condition_on_previous_text}
589
+
590
+ """
591
+
592
+ # Determine which files to return (trimmed or original)
593
+ return_mp3 = input_file
594
+ return_mp4 = video_file if video_file else None
595
+
596
+ if is_trimmed:
597
+ temp_dir_final_trimmed = os.path.join(os.getcwd(), "temp_output_final")
598
+ os.makedirs(temp_dir_final_trimmed, exist_ok=True)
599
+ temp_dirs.append(temp_dir_final_trimmed)
600
+
601
+ trimmed_mp3 = os.path.join(temp_dir_final_trimmed, f"{file_basename}_{ts}.mp3")
602
+ trimmed_video = None
603
+
604
+ if video_file:
605
+ file_ext = os.path.splitext(video_file)[1]
606
+ trimmed_video = os.path.join(temp_dir_final_trimmed, f"{file_basename}_{ts}{file_ext}")
607
+
608
+ duration_trim = end_time - start_time
609
+ cmd = [
610
+ 'ffmpeg',
611
+ '-ss', str(start_time),
612
+ '-i', video_file,
613
+ '-t', str(duration_trim),
614
+ '-c', 'copy',
615
+ '-avoid_negative_ts', 'make_zero',
616
+ trimmed_video,
617
+ '-y'
618
+ ]
619
+ subprocess.run(cmd, capture_output=True, check=True)
620
+
621
+ cmd_audio = [
622
+ 'ffmpeg',
623
+ '-i', trimmed_video,
624
+ '-vn',
625
+ '-acodec', 'libmp3lame',
626
+ '-q:a', '2',
627
+ trimmed_mp3,
628
+ '-y'
629
+ ]
630
+ subprocess.run(cmd_audio, capture_output=True, check=True)
631
+
632
+ return_mp3 = trimmed_mp3
633
+ return_mp4 = trimmed_video
634
+ else:
635
+ import shutil
636
+ shutil.copy2(audio_to_transcribe, trimmed_mp3)
637
+ return_mp3 = trimmed_mp3
638
+ return_mp4 = None
639
+
640
+ # Clear GPU cache after transcription
641
+ if torch.cuda.is_available():
642
+ torch.cuda.empty_cache()
643
+
644
+ # Final yield with all outputs
645
+ yield (
646
+ transcript_text,
647
+ txt_file,
648
+ srt_file,
649
+ vtt_file,
650
+ return_mp3,
651
+ return_mp4,
652
+ info_display,
653
+ all_segments_for_files,
654
+ actual_start_offset_for_transcription
655
+ )
656
+
657
+ except Exception as e:
658
+ import traceback
659
+ error_msg = f"❌ Lỗi transcription: {str(e)}\n\n{traceback.format_exc()}"
660
+
661
+ if torch.cuda.is_available():
662
+ torch.cuda.empty_cache()
663
+
664
+ yield (error_msg, None, None, None, None, None, None, None, 0)
665
+
666
+ # Function to clear outputs
667
+ def clear_outputs():
668
+ return (
669
+ "", # transcript_output
670
+ gr.update(value=None, visible=False), # source_mp3_output
671
+ gr.update(value=None, visible=False), # converted_mp3_output
672
+ gr.update(value=None, visible=False), # file_output
673
+ gr.update(value=None, visible=False), # srt_output
674
+ gr.update(value=None, visible=False), # vtt_output
675
+ gr.update(value=None, visible=False), # trimmed_mp3_output
676
+ gr.update(value=None, visible=False), # trimmed_mp4_output
677
+ gr.update(value=None, visible=False), # cut_zip_output
678
+ "", # statistics_output
679
+ "", # cut_status
680
+ None, # raw_segments_state
681
+ 0, # transcription_start_offset_state
682
+ gr.update(visible=False) # clear_output_btn
683
+ )
684
+
685
+
686
+ def cut_file_by_duration(audio_file, minutes_per_chunk):
687
+ """Cắt file mp3 thành các đoạn theo số phút"""
688
+ if not audio_file:
689
+ return [], "⚠️ Không có file audio để cắt."
690
+ try:
691
+ import datetime
692
+ chunk_secs = int(minutes_per_chunk) * 60
693
+ total_duration = get_duration(audio_file)
694
+ if total_duration <= chunk_secs:
695
+ return [], f"⚠️ File ngắn hơn {minutes_per_chunk} phút ({total_duration:.0f}s). Không cần cắt."
696
+ temp_dir = os.path.join(os.getcwd(), "temp_cut")
697
+ os.makedirs(temp_dir, exist_ok=True)
698
+ temp_dirs.append(temp_dir)
699
+ ts = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")
700
+ file_basename = os.path.splitext(os.path.basename(audio_file))[0]
701
+ output_files = []
702
+ part = 1
703
+ start = 0.0
704
+ while start < total_duration:
705
+ end = min(start + chunk_secs, total_duration)
706
+ out_path = os.path.join(temp_dir, f"{file_basename}_part{part:02d}_{ts}.mp3")
707
+ cmd = ['ffmpeg', '-ss', str(start), '-i', audio_file,
708
+ '-t', str(end - start), '-acodec', 'libmp3lame', '-q:a', '2', out_path, '-y']
709
+ result = subprocess.run(cmd, capture_output=True)
710
+ if result.returncode == 0:
711
+ output_files.append(out_path)
712
+ start += chunk_secs
713
+ part += 1
714
+ status = f"✅ Đã cắt thành {len(output_files)} đoạn × {minutes_per_chunk} phút"
715
+ return output_files, status
716
+ except Exception as e:
717
+ import traceback
718
+ return [], f"❌ Lỗi: {str(e)}\n{traceback.format_exc()}"
719
+
720
+
721
+ def cut_file_by_parts(audio_file, num_parts):
722
+ """Cắt file mp3 thành N phần bằng nhau"""
723
+ if not audio_file:
724
+ return [], "⚠️ Không có file audio để cắt."
725
+ try:
726
+ import datetime
727
+ num_parts = int(num_parts)
728
+ total_duration = get_duration(audio_file)
729
+ chunk_secs = total_duration / num_parts
730
+ temp_dir = os.path.join(os.getcwd(), "temp_cut")
731
+ os.makedirs(temp_dir, exist_ok=True)
732
+ temp_dirs.append(temp_dir)
733
+ ts = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")
734
+ file_basename = os.path.splitext(os.path.basename(audio_file))[0]
735
+ output_files = []
736
+ for part in range(1, num_parts + 1):
737
+ start = (part - 1) * chunk_secs
738
+ out_path = os.path.join(temp_dir, f"{file_basename}_part{part:02d}of{num_parts}_{ts}.mp3")
739
+ cmd = ['ffmpeg', '-ss', str(start), '-i', audio_file,
740
+ '-t', str(chunk_secs), '-acodec', 'libmp3lame', '-q:a', '2', out_path, '-y']
741
+ result = subprocess.run(cmd, capture_output=True)
742
+ if result.returncode == 0:
743
+ output_files.append(out_path)
744
+ status = f"✅ Đã cắt thành {len(output_files)} phần (~{chunk_secs/60:.1f} phút/phần)"
745
+ return output_files, status
746
+ except Exception as e:
747
+ import traceback
748
+ return [], f"❌ Lỗi: {str(e)}\n{traceback.format_exc()}"
749
+
750
+
751
+ def do_cut_file(audio_file, cut_mode, minutes_val, parts_val):
752
+ """Dispatcher: cắt theo phút hoặc theo phần, đóng gói zip"""
753
+ if not audio_file:
754
+ yield gr.update(value=None, visible=False), "⚠️ Không có file audio."
755
+ return
756
+ yield gr.update(visible=False), "⏳ Đang cắt file..."
757
+ if cut_mode == "Theo phút":
758
+ files, status = cut_file_by_duration(audio_file, minutes_val)
759
+ else:
760
+ files, status = cut_file_by_parts(audio_file, parts_val)
761
+ if not files:
762
+ yield gr.update(value=None, visible=False), status
763
+ return
764
+ import zipfile, datetime
765
+ ts = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")
766
+ file_basename = os.path.splitext(os.path.basename(audio_file))[0]
767
+ zip_dir = os.path.join(os.getcwd(), "temp_cut")
768
+ os.makedirs(zip_dir, exist_ok=True)
769
+ zip_path = os.path.join(zip_dir, f"{file_basename}_cut_{ts}.zip")
770
+ with zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zf:
771
+ for f in files:
772
+ zf.write(f, os.path.basename(f))
773
+ zip_size_mb = os.path.getsize(zip_path) / (1024 * 1024)
774
+ status += f"\n📦 ZIP: {os.path.basename(zip_path)} ({zip_size_mb:.1f} MB)"
775
+ yield gr.update(value=zip_path, visible=True), status
776
+
777
+
778
+ # ====== GRADIO INTERFACE ======
779
+
780
+ css = """
781
+ #textbox_id textarea {
782
+ color: black !important;
783
+ font-size: 16px !important;
784
+ font-family: 'IBM Plex Sans', sans-serif !important;
785
+ }
786
+
787
+ #textbox_id placeholder::textarea {
788
+ color: black !important;
789
+ font-size: 16px !important;
790
+ font-family: 'IBM Plex Sans', sans-serif !important;
791
+ }
792
+
793
+ #method_dropdown .menu button {
794
+ color: *primary_50 !important;
795
+ font-size: 16px !important;
796
+ }
797
+ """
798
+
799
+ theme = gr.themes.Default().set(
800
+ block_background_fill='*primary_50',
801
+ block_border_color='*button_primary_border_color',
802
+ block_label_text_color='*secondary_600',
803
+ block_info_text_color='*primary_700',
804
+ block_title_text_color='*primary_700',
805
+ body_text_size='*text_lg'
806
+ )
807
+
808
+ with gr.Blocks(theme=theme, css=css, title="Media Transcriber with Trimming") as demo:
809
+
810
+ gr.Markdown("# 🎤 Media Transcriber with Trimming")
811
+ gr.Markdown("Download từ YouTube (convert MP3) hoặc upload file, cắt (tùy chọn), và transcribe với faster-whisper")
812
+
813
+ with gr.Tab("Transcription"):
814
+ current_file = gr.State()
815
+ video_file_state = gr.State()
816
+ is_from_url = gr.State(False)
817
+ raw_segments_state = gr.State(None)
818
+ transcription_start_offset_state = gr.State(0)
819
+ transcribe_start_state = gr.State(0)
820
+ transcribe_end_state = gr.State(999999)
821
+
822
+ with gr.Row():
823
+ with gr.Column(scale=2):
824
+ url_input = gr.Textbox(
825
+ label="YouTube URL (tùy chọn)",
826
+ autoscroll=False,
827
+ elem_id="textbox_id",
828
+ lines=1,
829
+ max_lines=1,
830
+ placeholder="Nhập URL YouTube để tự động tải và convert sang MP3...",
831
+ )
832
+ upload_file = gr.File(
833
+ label="Upload file audio/video từ máy tính (.mp3 .wav .m4a .ogg .flac .aac .webm .mp4 .mkv .avi .mov .flv)",
834
+ file_types=["video", "audio", ".webm"],
835
+ type="filepath"
836
+ )
837
+ media_info = gr.Markdown(value="Nhập URL hoặc upload file để xem thông tin...")
838
+ output_preview = gr.Video(label="Preview (Media gốc)", height=300)
839
+
840
+ with gr.Row():
841
+ bitrate_dropdown = gr.Dropdown(
842
+ choices=["32k", "64k"],
843
+ value="32k",
844
+ label="🎚️ Bitrate",
845
+ info="32k: nhỏ hơn | 64k: tốt hơn một chút",
846
+ scale=1,
847
+ )
848
+ sample_rate_dropdown = gr.Dropdown(
849
+ choices=["8000", "16000"],
850
+ value="16000",
851
+ label="📊 Sample Rate (Hz)",
852
+ info="8kHz: tối thiểu | 16kHz: khuyến nghị cho transcribe",
853
+ scale=1,
854
+ )
855
+
856
+ convert_btn = gr.Button("🎵 Convert MP3", variant="secondary", size="lg")
857
+ convert_status = gr.Markdown(value="")
858
+
859
+ transcribe_btn = gr.Button("🎙️ Transcribe", variant="primary", size="lg", interactive=False)
860
+
861
+ with gr.Accordion("🔧 Phương thức & Timestamps", open=False):
862
+ with gr.Row():
863
+ method_dropdown = gr.Dropdown(
864
+ choices=["model.transcribe", "BatchedInferencePipeline"],
865
+ label="Phương thức",
866
+ scale=3,
867
+ elem_id="method_dropdown",
868
+ value="BatchedInferencePipeline",
869
+ info="model.transcribe: chất lượng cao | Batched: nhanh hơn"
870
+ )
871
+ beam_size_dropdown = gr.Dropdown(
872
+ choices=["1", "2", "3", "4", "5", "6", "7", "8", "9", "10"],
873
+ label="Beam Size",
874
+ scale=2,
875
+ elem_id="method_dropdown",
876
+ value="3",
877
+ info="Cao hơn = chính xác hơn nhưng chậm hơn"
878
+ )
879
+ include_timestamps_checkbox = gr.Checkbox(
880
+ label="Bao gồm Timestamps (HH:MM:SS)",
881
+ value=True,
882
+ info="Bao gồm timestamps trong kết quả transcription"
883
+ )
884
+
885
+ with gr.Accordion("⚙️ Tham số nâng cao (Batch & VAD)", open=False):
886
+ with gr.Row():
887
+ batch_size_slider = gr.Slider(
888
+ minimum=1,
889
+ maximum=64,
890
+ value=16,
891
+ step=1,
892
+ label="Batch Size (BatchedInferencePipeline)",
893
+ info="Số lượng đoạn âm thanh xử lý cùng lúc. Ảnh hưởng đến tốc độ và bộ nhớ GPU.",
894
+ visible=True
895
+ )
896
+ with gr.Row():
897
+ min_silence_duration_ms_slider = gr.Slider(
898
+ minimum=0,
899
+ maximum=2000,
900
+ value=500,
901
+ step=50,
902
+ label="VAD: Min Silence Duration (ms)",
903
+ info="Thời lượng im lặng tối thiểu để tách phân đoạn. Ảnh hưởng đến việc phát hiện câu/từ."
904
+ )
905
+ speech_pad_ms_slider = gr.Slider(
906
+ minimum=0,
907
+ maximum=1000,
908
+ value=400,
909
+ step=50,
910
+ label="VAD: Speech Pad (ms)",
911
+ info="Thêm thời gian vào đầu/cuối mỗi phân đoạn giọng nói. Giúp giữ lại bối cảnh."
912
+ )
913
+ with gr.Row():
914
+ no_speech_threshold_slider = gr.Slider(
915
+ minimum=0.0,
916
+ maximum=1.0,
917
+ value=0.55,
918
+ step=0.05,
919
+ label="VAD: No Speech Threshold",
920
+ info="Ngưỡng xác định khi nào không có lời nói."
921
+ )
922
+ condition_on_previous_text_checkbox = gr.Checkbox(
923
+ label="Condition on Previous Text",
924
+ value=False,
925
+ info="Sử dụng văn bản trước đó làm điều kiện để cải thiện tính nhất quán."
926
+ )
927
+ with gr.Row():
928
+ minimum_speech_duration_slider = gr.Slider(
929
+ minimum=0.0,
930
+ maximum=5.0,
931
+ value=0.1,
932
+ step=0.05,
933
+ label="VAD: Minimum Speech Duration (s)",
934
+ info="Thời lượng tối thiểu của một đoạn giọng nói. Giúp lọc các âm thanh ngắn, nhiễu."
935
+ )
936
+
937
+
938
+
939
+ with gr.Accordion("✂️ Cắt file MP3", open=False):
940
+ cut_mode_radio = gr.Radio(
941
+ choices=["Theo phút", "Theo phần"],
942
+ value="Theo phút",
943
+ label="Chế độ cắt",
944
+ info="Chỉ chọn một chế độ"
945
+ )
946
+ with gr.Row():
947
+ cut_minutes_dropdown = gr.Dropdown(
948
+ choices=["3", "5", "10", "15"],
949
+ value="5",
950
+ label="⏱️ Phút mỗi đoạn",
951
+ info="Áp dụng khi chọn Theo phút",
952
+ interactive=True,
953
+ scale=1,
954
+ )
955
+ cut_parts_dropdown = gr.Dropdown(
956
+ choices=["2", "3", "5", "10"],
957
+ value="2",
958
+ label="🔢 Số phần",
959
+ info="Áp dụng khi chọn Theo phần",
960
+ interactive=False,
961
+ scale=1,
962
+ )
963
+ cut_btn = gr.Button("✂️ Cut File", variant="secondary", size="lg")
964
+ cut_status = gr.Markdown(value="")
965
+
966
+
967
+ with gr.Column(scale=3):
968
+ placeholder_text = (
969
+ "📖 Hướng dẫn: Nhập Youtube video URL và bấm ENTER \n\n"
970
+ "📖 Hoặc: Upload file từ máy tính \n\n"
971
+ "💡Lưu ý:\n- Nếu không cắt file: -> transcribe toàn bộ.\n"
972
+ "- Nếu cắt file: -> chỉ transcribe phần được cắt."
973
+ )
974
+ transcript_output = gr.Textbox(
975
+ label="Kết quả Transcription",
976
+ lines=20,
977
+ interactive=True,
978
+ show_copy_button=True,
979
+ autoscroll=True,
980
+ elem_id="textbox_id",
981
+ placeholder=placeholder_text,
982
+ )
983
+ statistics_output = gr.Markdown(label="Thống kê Transcription", value="")
984
+ gr.Markdown("### 📥 Downloads")
985
+ source_mp3_output = gr.File(label="⬇️ MP3 từ Upload/URL", visible=False)
986
+ converted_mp3_output = gr.File(label="⬇️ MP3 đã Convert (Bitrate/Sample Rate)", visible=False)
987
+ with gr.Row():
988
+ file_output = gr.File(label="📄 Transcript (.txt)", visible=False)
989
+ srt_output = gr.File(label="📺 Subtitles (.srt)", visible=False)
990
+ with gr.Row():
991
+ vtt_output = gr.File(label="🌐 WebVTT (.vtt)", visible=False)
992
+ trimmed_mp3_output = gr.File(label="🎵 Audio MP3 (trimmed)", visible=False)
993
+ trimmed_mp4_output = gr.File(label="🎬 Video (trimmed - nếu có)", visible=False)
994
+ gr.Markdown("#### ✂️ File đã cắt")
995
+ cut_zip_output = gr.File(label="📦 Download ZIP các đoạn đã cắt", visible=False)
996
+
997
+ clear_output_btn = gr.Button("🧹 Xóa Kết quả", variant="secondary", visible=False)
998
+
999
+ # ====== EVENT HANDLERS ======
1000
+ url_input.submit(
1001
+ download_and_convert_to_mp3,
1002
+ inputs=[url_input],
1003
+ outputs=[current_file, video_file_state, media_info, transcribe_end_state, transcribe_start_state]
1004
+ ).then(
1005
+ lambda x: x,
1006
+ inputs=[video_file_state],
1007
+ outputs=[output_preview]
1008
+ ).then(
1009
+ lambda: True,
1010
+ outputs=[is_from_url]
1011
+ ).then(
1012
+ fn=lambda f: gr.update(value=f, visible=True) if f else gr.update(visible=False),
1013
+ inputs=[current_file],
1014
+ outputs=[source_mp3_output]
1015
+ )
1016
+
1017
+ upload_file.change(
1018
+ process_upload,
1019
+ inputs=[upload_file],
1020
+ outputs=[current_file, video_file_state, media_info, transcribe_end_state, transcribe_start_state]
1021
+ ).then(
1022
+ lambda x: x,
1023
+ inputs=[upload_file],
1024
+ outputs=[output_preview]
1025
+ ).then(
1026
+ lambda: False,
1027
+ outputs=[is_from_url]
1028
+ ).then(
1029
+ fn=lambda f: gr.update(value=f, visible=True) if f else gr.update(visible=False),
1030
+ inputs=[current_file],
1031
+ outputs=[source_mp3_output]
1032
+ )
1033
+
1034
+
1035
+
1036
+ # ====== NÚT CONVERT MP3 ======
1037
+ convert_btn.click(
1038
+ fn=lambda: (gr.update(interactive=False), gr.update(interactive=False), gr.update(value="⏳ Đang convert...")),
1039
+ inputs=None,
1040
+ outputs=[convert_btn, transcribe_btn, convert_status]
1041
+ ).then(
1042
+ fn=convert_to_mp3,
1043
+ inputs=[current_file, video_file_state, bitrate_dropdown, sample_rate_dropdown],
1044
+ outputs=[converted_mp3_output, convert_status]
1045
+ ).then(
1046
+ fn=lambda f: (gr.update(interactive=True), gr.update(interactive=True), gr.update(visible=True) if f else gr.update(visible=False)),
1047
+ inputs=[converted_mp3_output],
1048
+ outputs=[convert_btn, transcribe_btn, converted_mp3_output]
1049
+ )
1050
+
1051
+ # ====== NÚT TRANSCRIBE - Disable Trim button khi đang transcribe ======
1052
+ transcribe_btn.click(
1053
+ fn=lambda: (gr.update(interactive=False), gr.update(interactive=False)),
1054
+ inputs=None,
1055
+ outputs=[transcribe_btn, convert_btn]
1056
+ ).then(
1057
+ fn=transcribe_audio,
1058
+ inputs=[
1059
+ current_file,
1060
+ video_file_state,
1061
+ method_dropdown,
1062
+ beam_size_dropdown,
1063
+ transcribe_start_state,
1064
+ transcribe_end_state,
1065
+ include_timestamps_checkbox,
1066
+ batch_size_slider,
1067
+ min_silence_duration_ms_slider,
1068
+ speech_pad_ms_slider,
1069
+ no_speech_threshold_slider,
1070
+ condition_on_previous_text_checkbox,
1071
+ minimum_speech_duration_slider
1072
+ ],
1073
+ outputs=[
1074
+ transcript_output,
1075
+ file_output,
1076
+ srt_output,
1077
+ vtt_output,
1078
+ trimmed_mp3_output,
1079
+ trimmed_mp4_output,
1080
+ statistics_output,
1081
+ raw_segments_state,
1082
+ transcription_start_offset_state
1083
+ ]
1084
+ ).then(
1085
+ fn=lambda: (gr.update(interactive=True), gr.update(interactive=True)),
1086
+ inputs=None,
1087
+ outputs=[transcribe_btn, convert_btn]
1088
+ ).then(
1089
+ fn=lambda mp4: (
1090
+ gr.update(visible=True), gr.update(visible=True), gr.update(visible=True),
1091
+ gr.update(visible=True),
1092
+ gr.update(visible=True) if mp4 else gr.update(visible=False),
1093
+ gr.update(visible=True)
1094
+ ),
1095
+ inputs=[trimmed_mp4_output],
1096
+ outputs=[
1097
+ file_output, srt_output, vtt_output,
1098
+ trimmed_mp3_output, trimmed_mp4_output,
1099
+ clear_output_btn
1100
+ ]
1101
+ )
1102
+
1103
+ # ====== CUT FILE - Radio toggle interactivity ======
1104
+ cut_mode_radio.change(
1105
+ fn=lambda mode: (
1106
+ gr.update(interactive=(mode == "Theo phút")),
1107
+ gr.update(interactive=(mode == "Theo phần"))
1108
+ ),
1109
+ inputs=[cut_mode_radio],
1110
+ outputs=[cut_minutes_dropdown, cut_parts_dropdown]
1111
+ )
1112
+
1113
+ cut_btn.click(
1114
+ fn=lambda: gr.update(interactive=False),
1115
+ inputs=None,
1116
+ outputs=[cut_btn]
1117
+ ).then(
1118
+ fn=do_cut_file,
1119
+ inputs=[current_file, cut_mode_radio, cut_minutes_dropdown, cut_parts_dropdown],
1120
+ outputs=[cut_zip_output, cut_status]
1121
+ ).then(
1122
+ fn=lambda: gr.update(interactive=True),
1123
+ inputs=None,
1124
+ outputs=[cut_btn]
1125
+ )
1126
+
1127
+ # Clear Output button handler
1128
+ clear_output_btn.click(
1129
+ fn=clear_outputs,
1130
+ inputs=None,
1131
+ outputs=[
1132
+ transcript_output,
1133
+ source_mp3_output,
1134
+ converted_mp3_output,
1135
+ file_output,
1136
+ srt_output,
1137
+ vtt_output,
1138
+ trimmed_mp3_output,
1139
+ trimmed_mp4_output,
1140
+ cut_zip_output,
1141
+ statistics_output,
1142
+ cut_status,
1143
+ raw_segments_state,
1144
+ transcription_start_offset_state,
1145
+ clear_output_btn
1146
+ ]
1147
+ )
1148
+
1149
+ # Dynamic timestamp toggle handler
1150
+ include_timestamps_checkbox.change(
1151
+ fn=format_transcript_display,
1152
+ inputs=[raw_segments_state, include_timestamps_checkbox, transcription_start_offset_state],
1153
+ outputs=[transcript_output]
1154
+ )
1155
+
1156
+ # Event for method_dropdown to control batch_size_slider visibility
1157
+ method_dropdown.change(
1158
+ fn=lambda m: gr.update(visible=(m == "BatchedInferencePipeline")),
1159
+ inputs=[method_dropdown],
1160
+ outputs=[batch_size_slider]
1161
+ )
1162
+
1163
+ # ====== QUEUE + LAUNCH ======
1164
+ demo.queue(default_concurrency_limit=2)
1165
+ demo.launch()
requirements.txt ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ gradio
2
+ faster-whisper
3
+ yt-dlp
4
+ moviepy
5
+ pydub
6
+ torch