Spaces:
Runtime error
Runtime error
| # app.py | |
| import os, tempfile, uuid, math, random | |
| from pathlib import Path | |
| from io import BytesIO | |
| import numpy as np | |
| from PIL import Image, ImageDraw | |
| import gradio as gr | |
| import moviepy.editor as mpy | |
| from pydub import AudioSegment | |
| # Try faster_whisper first, fallback to whisper | |
| WHISPER_AVAILABLE = False | |
| try: | |
| from faster_whisper import WhisperModel | |
| whisper_model = WhisperModel("small", device="cpu", compute_type="int8") | |
| WHISPER_AVAILABLE = True | |
| except Exception: | |
| import whisper | |
| whisper_model = whisper.load_model("small") | |
| # Small instruction model (Flan-T5 small) | |
| from transformers import AutoTokenizer, AutoModelForSeq2SeqLM, pipeline | |
| MODEL_NAME = "google/flan-t5-small" | |
| tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME) | |
| model = AutoModelForSeq2SeqLM.from_pretrained(MODEL_NAME) | |
| llm = pipeline("text2text-generation", model=model, tokenizer=tokenizer) | |
| # gTTS for simple female voice | |
| from gtts import gTTS | |
| # ----------------- Helpers ----------------- | |
| def transcribe_audio(path, lang=None): | |
| try: | |
| if WHISPER_AVAILABLE: | |
| segments, _ = whisper_model.transcribe(path, language=lang) if lang else whisper_model.transcribe(path) | |
| return " ".join([s.text for s in segments]) | |
| else: | |
| res = whisper_model.transcribe(path, language=lang) if lang else whisper_model.transcribe(path) | |
| return res["text"] | |
| except Exception as e: | |
| return "" | |
| def ask_llm(user_text): | |
| prompt = f"You are a friendly helpful tutor and assistant. Reply concisely and kindly. Also ask one follow-up question when relevant.\nUser: {user_text}\nAssistant:" | |
| out = llm(prompt, max_length=180, do_sample=False) | |
| return out[0]["generated_text"] | |
| def tts_gtts(text, out_path, lang="en"): | |
| if not text.strip(): | |
| # tiny silent mp3 | |
| silent = AudioSegment.silent(duration=400) | |
| silent.export(out_path, format="mp3") | |
| return out_path | |
| tts = gTTS(text=text, lang=lang, slow=False) | |
| tts.save(out_path) | |
| return out_path | |
| def compute_envelope(audio_path, fps=25): | |
| seg = AudioSegment.from_file(audio_path) | |
| samples = np.array(seg.get_array_of_samples()).astype(np.float32) | |
| if seg.channels > 1: | |
| samples = samples.reshape((-1, seg.channels)).mean(axis=1) | |
| if samples.size == 0: | |
| return np.zeros(1) | |
| samples = samples / (np.max(np.abs(samples)) + 1e-9) | |
| duration = seg.duration_seconds | |
| n_frames = max(1, int(duration * fps)) | |
| parts = np.array_split(samples, n_frames) | |
| env = np.array([np.sqrt(np.mean(p**2)) if p.size>0 else 0 for p in parts]) | |
| # normalize 0..1 | |
| env = (env - env.min()) / (env.max() - env.min() + 1e-9) | |
| return env | |
| def create_talking_clip(image_pil, audio_path, out_video_path, fps=25, emotion="neutral"): | |
| envelope = compute_envelope(audio_path, fps=fps) | |
| duration = max(0.5, len(envelope) / fps) | |
| w,h = image_pil.size | |
| # jaw area estimate | |
| jw = int(w * 0.26); jh = int(h * 0.08) | |
| jx = int(w*0.5 - jw/2); jy = int(h*0.68 - jh/2) | |
| # emotion-driven small head tilt function | |
| if emotion == "happy": | |
| tilt_fn = lambda t: math.sin(2*math.pi*t/duration)*2.2 | |
| elif emotion == "thinking": | |
| tilt_fn = lambda t: math.sin(2*math.pi*t/duration)*-2.5 | |
| elif emotion == "surprised": | |
| tilt_fn = lambda t: math.sin(2*math.pi*t/duration)*1.8 | |
| else: | |
| tilt_fn = lambda t: math.sin(2*math.pi*t/duration)*0.6 | |
| def make_frame(t): | |
| i = min(int(t*fps), len(envelope)-1) | |
| level = float(envelope[i]) | |
| frame = image_pil.copy().convert("RGBA") | |
| draw = ImageDraw.Draw(frame, 'RGBA') | |
| # head tilt (rotate slightly) | |
| angle = tilt_fn(t) | |
| frame = frame.rotate(angle, resample=Image.BICUBIC, center=(w//2, h//3), expand=False) | |
| # mouth ellipse overlay to simulate opening | |
| mouth_h = int(jh * (1.0 + level*1.2)) | |
| mouth_y = int(jy + jh - mouth_h/2) | |
| alpha = int(20 + level*120) | |
| draw.ellipse([jx, mouth_y, jx+jw, mouth_y+mouth_h], fill=(10,10,10, alpha)) | |
| # occasional blink | |
| if (int(t*2) % 7) == 0 and random.random() > 0.6: | |
| draw.rectangle([0, 0, w, int(h*0.23)], fill=(245,245,255,230)) | |
| return np.asarray(frame) | |
| clip = mpy.VideoClip(make_frame, duration=duration) | |
| audio = mpy.AudioFileClip(audio_path) | |
| clip = clip.set_audio(audio) | |
| clip.write_videofile(out_video_path, fps=fps, codec="libx264", audio_codec="aac", verbose=False, logger=None) | |
| return out_video_path | |
| def detect_emotion_from_text(text): | |
| t = text.lower() | |
| if any(w in t for w in ["happy","love","great","good","awesome"]): | |
| return "happy" | |
| if any(w in t for w in ["why","how","think","confused","wonder"]): | |
| return "thinking" | |
| if any(w in t for w in ["wow","surprise","amazed","shocked"]): | |
| return "surprised" | |
| if any(w in t for w in ["sorry","shy","nervous"]): | |
| return "shy" | |
| return "neutral" | |
| # ----------------- Gradio interface ----------------- | |
| def process(image, upload_audio, mic_audio, typed_text): | |
| uid = str(uuid.uuid4())[:8] | |
| tmp = Path(tempfile.gettempdir()) / f"avatar_{uid}" | |
| tmp.mkdir(parents=True, exist_ok=True) | |
| if image is None: | |
| return None, "Please upload an avatar image (head+shoulders).", None | |
| # normalize image | |
| if isinstance(image, np.ndarray): | |
| image_pil = Image.fromarray(image).convert("RGBA") | |
| else: | |
| image_pil = Image.open(image).convert("RGBA") | |
| # get user text (typed or transcribed) | |
| user_text = "" | |
| audio_in_path = None | |
| if typed_text and typed_text.strip(): | |
| user_text = typed_text.strip() | |
| lang_hint = "en" | |
| else: | |
| # priority: mic_audio -> upload_audio | |
| audio_file = mic_audio if mic_audio else upload_audio | |
| if not audio_file: | |
| return None, "No audio or text provided. Speak or type.", None | |
| audio_path = tmp / "user.wav" | |
| with open(audio_path, "wb") as f: | |
| f.write(audio_file.read()) | |
| audio_in_path = str(audio_path) | |
| user_text = transcribe_audio(str(audio_path)) | |
| lang_hint = "en" | |
| if not user_text: | |
| return None, "Couldn't transcribe. Try again or type.", None | |
| # get LLM reply | |
| reply_text = ask_llm(user_text) | |
| # detect emotion for gestures | |
| emotion = detect_emotion_from_text(user_text + " " + reply_text) | |
| # TTS generate reply audio | |
| tts_path = tmp / "reply.mp3" | |
| tts_gtts(reply_text, str(tts_path), lang="en") | |
| # produce talking clip (avatar + mouth animation) | |
| out_video = tmp / "talking.mp4" | |
| create_talking_clip(image_pil, str(tts_path), str(out_video), fps=25, emotion=emotion) | |
| return str(out_video), reply_text, str(tts_path) | |
| # Gradio UI | |
| title = "Live Animated Avatar Companion (Free)" | |
| desc = "Upload an avatar image (head+shoulders). Speak or type. The app transcribes, replies, synthesizes voice, and produces a talking video with mouth animation, blink & tilt." | |
| demo = gr.Interface( | |
| fn=process, | |
| inputs=[ | |
| gr.Image(type="pil", label="Upload avatar (head & shoulders PNG)"), | |
| gr.Audio(source="upload", type="file", label="Upload audio (optional)"), | |
| gr.Audio(source="microphone", type="file", label="Record via mic (optional)"), | |
| gr.Textbox(lines=2, placeholder="Or type your message (optional)", label="Type message") | |
| ], | |
| outputs=[ | |
| gr.Video(label="Talking clip (MP4)"), | |
| gr.Textbox(label="Assistant reply"), | |
| gr.Audio(label="Reply audio (mp3)", type="file") | |
| ], | |
| title=title, | |
| description=desc, | |
| allow_flagging="never", | |
| ) | |
| if __name__ == "__main__": | |
| demo.launch() |