File size: 7,654 Bytes
b8fa61c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
# app.py
import os, tempfile, uuid, math, random
from pathlib import Path
from io import BytesIO
import numpy as np
from PIL import Image, ImageDraw
import gradio as gr
import moviepy.editor as mpy
from pydub import AudioSegment

# Try faster_whisper first, fallback to whisper
WHISPER_AVAILABLE = False
try:
    from faster_whisper import WhisperModel
    whisper_model = WhisperModel("small", device="cpu", compute_type="int8")
    WHISPER_AVAILABLE = True
except Exception:
    import whisper
    whisper_model = whisper.load_model("small")

# Small instruction model (Flan-T5 small)
from transformers import AutoTokenizer, AutoModelForSeq2SeqLM, pipeline
MODEL_NAME = "google/flan-t5-small"
tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)
model = AutoModelForSeq2SeqLM.from_pretrained(MODEL_NAME)
llm = pipeline("text2text-generation", model=model, tokenizer=tokenizer)

# gTTS for simple female voice
from gtts import gTTS

# ----------------- Helpers -----------------
def transcribe_audio(path, lang=None):
    try:
        if WHISPER_AVAILABLE:
            segments, _ = whisper_model.transcribe(path, language=lang) if lang else whisper_model.transcribe(path)
            return " ".join([s.text for s in segments])
        else:
            res = whisper_model.transcribe(path, language=lang) if lang else whisper_model.transcribe(path)
            return res["text"]
    except Exception as e:
        return ""

def ask_llm(user_text):
    prompt = f"You are a friendly helpful tutor and assistant. Reply concisely and kindly. Also ask one follow-up question when relevant.\nUser: {user_text}\nAssistant:"
    out = llm(prompt, max_length=180, do_sample=False)
    return out[0]["generated_text"]

def tts_gtts(text, out_path, lang="en"):
    if not text.strip():
        # tiny silent mp3
        silent = AudioSegment.silent(duration=400)
        silent.export(out_path, format="mp3")
        return out_path
    tts = gTTS(text=text, lang=lang, slow=False)
    tts.save(out_path)
    return out_path

def compute_envelope(audio_path, fps=25):
    seg = AudioSegment.from_file(audio_path)
    samples = np.array(seg.get_array_of_samples()).astype(np.float32)
    if seg.channels > 1:
        samples = samples.reshape((-1, seg.channels)).mean(axis=1)
    if samples.size == 0:
        return np.zeros(1)
    samples = samples / (np.max(np.abs(samples)) + 1e-9)
    duration = seg.duration_seconds
    n_frames = max(1, int(duration * fps))
    parts = np.array_split(samples, n_frames)
    env = np.array([np.sqrt(np.mean(p**2)) if p.size>0 else 0 for p in parts])
    # normalize 0..1
    env = (env - env.min()) / (env.max() - env.min() + 1e-9)
    return env

def create_talking_clip(image_pil, audio_path, out_video_path, fps=25, emotion="neutral"):
    envelope = compute_envelope(audio_path, fps=fps)
    duration = max(0.5, len(envelope) / fps)
    w,h = image_pil.size

    # jaw area estimate
    jw = int(w * 0.26); jh = int(h * 0.08)
    jx = int(w*0.5 - jw/2); jy = int(h*0.68 - jh/2)

    # emotion-driven small head tilt function
    if emotion == "happy":
        tilt_fn = lambda t: math.sin(2*math.pi*t/duration)*2.2
    elif emotion == "thinking":
        tilt_fn = lambda t: math.sin(2*math.pi*t/duration)*-2.5
    elif emotion == "surprised":
        tilt_fn = lambda t: math.sin(2*math.pi*t/duration)*1.8
    else:
        tilt_fn = lambda t: math.sin(2*math.pi*t/duration)*0.6

    def make_frame(t):
        i = min(int(t*fps), len(envelope)-1)
        level = float(envelope[i])
        frame = image_pil.copy().convert("RGBA")
        draw = ImageDraw.Draw(frame, 'RGBA')

        # head tilt (rotate slightly)
        angle = tilt_fn(t)
        frame = frame.rotate(angle, resample=Image.BICUBIC, center=(w//2, h//3), expand=False)

        # mouth ellipse overlay to simulate opening
        mouth_h = int(jh * (1.0 + level*1.2))
        mouth_y = int(jy + jh - mouth_h/2)
        alpha = int(20 + level*120)
        draw.ellipse([jx, mouth_y, jx+jw, mouth_y+mouth_h], fill=(10,10,10, alpha))

        # occasional blink
        if (int(t*2) % 7) == 0 and random.random() > 0.6:
            draw.rectangle([0, 0, w, int(h*0.23)], fill=(245,245,255,230))

        return np.asarray(frame)

    clip = mpy.VideoClip(make_frame, duration=duration)
    audio = mpy.AudioFileClip(audio_path)
    clip = clip.set_audio(audio)
    clip.write_videofile(out_video_path, fps=fps, codec="libx264", audio_codec="aac", verbose=False, logger=None)
    return out_video_path

def detect_emotion_from_text(text):
    t = text.lower()
    if any(w in t for w in ["happy","love","great","good","awesome"]):
        return "happy"
    if any(w in t for w in ["why","how","think","confused","wonder"]):
        return "thinking"
    if any(w in t for w in ["wow","surprise","amazed","shocked"]):
        return "surprised"
    if any(w in t for w in ["sorry","shy","nervous"]):
        return "shy"
    return "neutral"

# ----------------- Gradio interface -----------------
def process(image, upload_audio, mic_audio, typed_text):
    uid = str(uuid.uuid4())[:8]
    tmp = Path(tempfile.gettempdir()) / f"avatar_{uid}"
    tmp.mkdir(parents=True, exist_ok=True)

    if image is None:
        return None, "Please upload an avatar image (head+shoulders).", None

    # normalize image
    if isinstance(image, np.ndarray):
        image_pil = Image.fromarray(image).convert("RGBA")
    else:
        image_pil = Image.open(image).convert("RGBA")

    # get user text (typed or transcribed)
    user_text = ""
    audio_in_path = None
    if typed_text and typed_text.strip():
        user_text = typed_text.strip()
        lang_hint = "en"
    else:
        # priority: mic_audio -> upload_audio
        audio_file = mic_audio if mic_audio else upload_audio
        if not audio_file:
            return None, "No audio or text provided. Speak or type.", None
        audio_path = tmp / "user.wav"
        with open(audio_path, "wb") as f:
            f.write(audio_file.read())
        audio_in_path = str(audio_path)
        user_text = transcribe_audio(str(audio_path))
        lang_hint = "en"

    if not user_text:
        return None, "Couldn't transcribe. Try again or type.", None

    # get LLM reply
    reply_text = ask_llm(user_text)

    # detect emotion for gestures
    emotion = detect_emotion_from_text(user_text + " " + reply_text)

    # TTS generate reply audio
    tts_path = tmp / "reply.mp3"
    tts_gtts(reply_text, str(tts_path), lang="en")

    # produce talking clip (avatar + mouth animation)
    out_video = tmp / "talking.mp4"
    create_talking_clip(image_pil, str(tts_path), str(out_video), fps=25, emotion=emotion)

    return str(out_video), reply_text, str(tts_path)

# Gradio UI
title = "Live Animated Avatar Companion (Free)"
desc = "Upload an avatar image (head+shoulders). Speak or type. The app transcribes, replies, synthesizes voice, and produces a talking video with mouth animation, blink & tilt."

demo = gr.Interface(
    fn=process,
    inputs=[
        gr.Image(type="pil", label="Upload avatar (head & shoulders PNG)"),
        gr.Audio(source="upload", type="file", label="Upload audio (optional)"),
        gr.Audio(source="microphone", type="file", label="Record via mic (optional)"),
        gr.Textbox(lines=2, placeholder="Or type your message (optional)", label="Type message")
    ],
    outputs=[
        gr.Video(label="Talking clip (MP4)"),
        gr.Textbox(label="Assistant reply"),
        gr.Audio(label="Reply audio (mp3)", type="file")
    ],
    title=title,
    description=desc,
    allow_flagging="never",
)

if __name__ == "__main__":
    demo.launch()