Spaces:
Runtime error
Runtime error
File size: 7,654 Bytes
b8fa61c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 | # app.py
import os, tempfile, uuid, math, random
from pathlib import Path
from io import BytesIO
import numpy as np
from PIL import Image, ImageDraw
import gradio as gr
import moviepy.editor as mpy
from pydub import AudioSegment
# Try faster_whisper first, fallback to whisper
WHISPER_AVAILABLE = False
try:
from faster_whisper import WhisperModel
whisper_model = WhisperModel("small", device="cpu", compute_type="int8")
WHISPER_AVAILABLE = True
except Exception:
import whisper
whisper_model = whisper.load_model("small")
# Small instruction model (Flan-T5 small)
from transformers import AutoTokenizer, AutoModelForSeq2SeqLM, pipeline
MODEL_NAME = "google/flan-t5-small"
tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)
model = AutoModelForSeq2SeqLM.from_pretrained(MODEL_NAME)
llm = pipeline("text2text-generation", model=model, tokenizer=tokenizer)
# gTTS for simple female voice
from gtts import gTTS
# ----------------- Helpers -----------------
def transcribe_audio(path, lang=None):
try:
if WHISPER_AVAILABLE:
segments, _ = whisper_model.transcribe(path, language=lang) if lang else whisper_model.transcribe(path)
return " ".join([s.text for s in segments])
else:
res = whisper_model.transcribe(path, language=lang) if lang else whisper_model.transcribe(path)
return res["text"]
except Exception as e:
return ""
def ask_llm(user_text):
prompt = f"You are a friendly helpful tutor and assistant. Reply concisely and kindly. Also ask one follow-up question when relevant.\nUser: {user_text}\nAssistant:"
out = llm(prompt, max_length=180, do_sample=False)
return out[0]["generated_text"]
def tts_gtts(text, out_path, lang="en"):
if not text.strip():
# tiny silent mp3
silent = AudioSegment.silent(duration=400)
silent.export(out_path, format="mp3")
return out_path
tts = gTTS(text=text, lang=lang, slow=False)
tts.save(out_path)
return out_path
def compute_envelope(audio_path, fps=25):
seg = AudioSegment.from_file(audio_path)
samples = np.array(seg.get_array_of_samples()).astype(np.float32)
if seg.channels > 1:
samples = samples.reshape((-1, seg.channels)).mean(axis=1)
if samples.size == 0:
return np.zeros(1)
samples = samples / (np.max(np.abs(samples)) + 1e-9)
duration = seg.duration_seconds
n_frames = max(1, int(duration * fps))
parts = np.array_split(samples, n_frames)
env = np.array([np.sqrt(np.mean(p**2)) if p.size>0 else 0 for p in parts])
# normalize 0..1
env = (env - env.min()) / (env.max() - env.min() + 1e-9)
return env
def create_talking_clip(image_pil, audio_path, out_video_path, fps=25, emotion="neutral"):
envelope = compute_envelope(audio_path, fps=fps)
duration = max(0.5, len(envelope) / fps)
w,h = image_pil.size
# jaw area estimate
jw = int(w * 0.26); jh = int(h * 0.08)
jx = int(w*0.5 - jw/2); jy = int(h*0.68 - jh/2)
# emotion-driven small head tilt function
if emotion == "happy":
tilt_fn = lambda t: math.sin(2*math.pi*t/duration)*2.2
elif emotion == "thinking":
tilt_fn = lambda t: math.sin(2*math.pi*t/duration)*-2.5
elif emotion == "surprised":
tilt_fn = lambda t: math.sin(2*math.pi*t/duration)*1.8
else:
tilt_fn = lambda t: math.sin(2*math.pi*t/duration)*0.6
def make_frame(t):
i = min(int(t*fps), len(envelope)-1)
level = float(envelope[i])
frame = image_pil.copy().convert("RGBA")
draw = ImageDraw.Draw(frame, 'RGBA')
# head tilt (rotate slightly)
angle = tilt_fn(t)
frame = frame.rotate(angle, resample=Image.BICUBIC, center=(w//2, h//3), expand=False)
# mouth ellipse overlay to simulate opening
mouth_h = int(jh * (1.0 + level*1.2))
mouth_y = int(jy + jh - mouth_h/2)
alpha = int(20 + level*120)
draw.ellipse([jx, mouth_y, jx+jw, mouth_y+mouth_h], fill=(10,10,10, alpha))
# occasional blink
if (int(t*2) % 7) == 0 and random.random() > 0.6:
draw.rectangle([0, 0, w, int(h*0.23)], fill=(245,245,255,230))
return np.asarray(frame)
clip = mpy.VideoClip(make_frame, duration=duration)
audio = mpy.AudioFileClip(audio_path)
clip = clip.set_audio(audio)
clip.write_videofile(out_video_path, fps=fps, codec="libx264", audio_codec="aac", verbose=False, logger=None)
return out_video_path
def detect_emotion_from_text(text):
t = text.lower()
if any(w in t for w in ["happy","love","great","good","awesome"]):
return "happy"
if any(w in t for w in ["why","how","think","confused","wonder"]):
return "thinking"
if any(w in t for w in ["wow","surprise","amazed","shocked"]):
return "surprised"
if any(w in t for w in ["sorry","shy","nervous"]):
return "shy"
return "neutral"
# ----------------- Gradio interface -----------------
def process(image, upload_audio, mic_audio, typed_text):
uid = str(uuid.uuid4())[:8]
tmp = Path(tempfile.gettempdir()) / f"avatar_{uid}"
tmp.mkdir(parents=True, exist_ok=True)
if image is None:
return None, "Please upload an avatar image (head+shoulders).", None
# normalize image
if isinstance(image, np.ndarray):
image_pil = Image.fromarray(image).convert("RGBA")
else:
image_pil = Image.open(image).convert("RGBA")
# get user text (typed or transcribed)
user_text = ""
audio_in_path = None
if typed_text and typed_text.strip():
user_text = typed_text.strip()
lang_hint = "en"
else:
# priority: mic_audio -> upload_audio
audio_file = mic_audio if mic_audio else upload_audio
if not audio_file:
return None, "No audio or text provided. Speak or type.", None
audio_path = tmp / "user.wav"
with open(audio_path, "wb") as f:
f.write(audio_file.read())
audio_in_path = str(audio_path)
user_text = transcribe_audio(str(audio_path))
lang_hint = "en"
if not user_text:
return None, "Couldn't transcribe. Try again or type.", None
# get LLM reply
reply_text = ask_llm(user_text)
# detect emotion for gestures
emotion = detect_emotion_from_text(user_text + " " + reply_text)
# TTS generate reply audio
tts_path = tmp / "reply.mp3"
tts_gtts(reply_text, str(tts_path), lang="en")
# produce talking clip (avatar + mouth animation)
out_video = tmp / "talking.mp4"
create_talking_clip(image_pil, str(tts_path), str(out_video), fps=25, emotion=emotion)
return str(out_video), reply_text, str(tts_path)
# Gradio UI
title = "Live Animated Avatar Companion (Free)"
desc = "Upload an avatar image (head+shoulders). Speak or type. The app transcribes, replies, synthesizes voice, and produces a talking video with mouth animation, blink & tilt."
demo = gr.Interface(
fn=process,
inputs=[
gr.Image(type="pil", label="Upload avatar (head & shoulders PNG)"),
gr.Audio(source="upload", type="file", label="Upload audio (optional)"),
gr.Audio(source="microphone", type="file", label="Record via mic (optional)"),
gr.Textbox(lines=2, placeholder="Or type your message (optional)", label="Type message")
],
outputs=[
gr.Video(label="Talking clip (MP4)"),
gr.Textbox(label="Assistant reply"),
gr.Audio(label="Reply audio (mp3)", type="file")
],
title=title,
description=desc,
allow_flagging="never",
)
if __name__ == "__main__":
demo.launch() |