File size: 14,627 Bytes
7dc760b
 
 
dd09aa6
 
 
 
 
 
4978487
dd09aa6
 
 
7feb18b
 
 
 
 
7dc760b
4978487
 
 
 
 
 
 
50afb76
dd09aa6
 
 
d2e57f7
7dc760b
4978487
d2e57f7
7dc760b
dd09aa6
4978487
 
dd09aa6
 
4978487
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7dc760b
 
 
50afb76
7dc760b
50afb76
7dc760b
 
 
7feb18b
 
 
 
 
77fdb5a
dd09aa6
 
 
 
 
 
 
 
77fdb5a
dd09aa6
 
 
 
 
4978487
dd09aa6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
77fdb5a
dd09aa6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7dc760b
40edffa
7dc760b
4978487
 
40edffa
 
7dc760b
4978487
 
 
7dc760b
4978487
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
dd09aa6
 
4978487
dd09aa6
 
4978487
dd09aa6
7dc760b
dd09aa6
 
 
 
 
 
 
 
 
 
d2e57f7
 
4978487
dd09aa6
 
 
 
 
 
4978487
dd09aa6
 
 
 
 
 
 
 
4978487
dd09aa6
7dc760b
dd09aa6
 
4978487
dd09aa6
 
 
 
 
 
 
7dc760b
dd09aa6
 
4978487
dd09aa6
 
 
 
7dc760b
dd09aa6
 
4978487
dd09aa6
 
 
 
4978487
dd09aa6
 
4978487
dd09aa6
 
 
 
 
4978487
dd09aa6
 
 
 
4978487
dd09aa6
 
 
4978487
7dc760b
77fdb5a
dd09aa6
d2e57f7
77fdb5a
4978487
77fdb5a
 
 
4978487
d2e57f7
77fdb5a
 
 
4978487
 
77fdb5a
 
 
4978487
77fdb5a
 
 
 
4978487
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
77fdb5a
4978487
 
77fdb5a
 
 
4978487
77fdb5a
 
 
 
 
 
4978487
77fdb5a
 
4978487
 
 
 
77fdb5a
 
 
 
 
 
 
 
 
40edffa
4978487
77fdb5a
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
import gradio as gr
import os
import time
import requests
import base64
import jwt
import io
from PIL import Image

# Пытаемся импортировать Google GenAI
try:
    from google import genai
    from google.genai import types
    HAS_GENAI = True
    GENAI_ERR = ""
except ImportError as e:
    HAS_GENAI = False
    GENAI_ERR = str(e)

# Пытаемся импортировать OpenAI
try:
    from openai import OpenAI
    HAS_OPENAI = True
except ImportError:
    HAS_OPENAI = False

# --- SECRETS AND CONFIGURATION ---
GOOGLE_API_KEY = os.environ.get("GOOGLE_API_KEY") 
KLING_ACCESS_KEY = os.getenv("KLING_ACCESS_KEY")
KLING_SECRET_KEY = os.getenv("KLING_SECRET_KEY")
OPENAI_API_KEY = os.getenv("OPENAI_API_KEY")

# --- STATIC PROMPT FOR PHOTO ---
SYSTEM_PROMPT = "Hyper-realistic Instagram influencer lifestyle photo, shot on iPhone 15 Pro. The exact person[refer to subject images] is looking directly at the camera with a warm, highly attractive, and confident smile showing natural bright white teeth. Healthy, clear, highly realistic skin texture (natural, not overly airbrushed, but free of blemishes or rough spots, with a soft flattering glow). They are casually holding the product [refer to product images] close to the camera, showing it to the viewers. Flattering soft daylight from a window combined with a subtle ambient glow on the face. The background is a stylish, slightly eclectic modern loft with unique decor, lush indoor plants, and rich textures, giving a lived-in, creative, and non-generic vibe. Authentic social media blogger aesthetic, exact facial likeness, highly detailed, and visually pleasing."

def generate_kling_token(ak, sk):
    headers = {"alg": "HS256", "typ": "JWT"}
    payload = {"iss": ak, "exp": int(time.time()) + 1800, "nbf": int(time.time()) - 5}
    return jwt.encode(payload, sk, algorithm="HS256", headers=headers)

# Функция для вызова GPT-4o Vision
def generate_video_prompt_with_gpt(base64_image_for_gpt):
    if not HAS_OPENAI or not OPENAI_API_KEY:
        return "Natural handheld UGC-style video. The influencer smiles warmly, showing the product to the camera. Subtle micro-shakes, photorealistic, cinematic lighting, casual authentic vibe."
    
    try:
        client = OpenAI(api_key=OPENAI_API_KEY)
        response = client.chat.completions.create(
            model="gpt-4o",
            messages=[
                {
                    "role": "system",
                    "content": "You are an expert AI video generation prompter. Your goal is to write a short, highly detailed prompt for Kling AI (Image-to-Video model). Output ONLY the prompt text, nothing else."
                },
                {
                    "role": "user",
                    "content":[
                        {
                            "type": "text", 
                            "text": "Analyze this starting frame. Write a prompt to animate it. Rules:\n1. Keep it under 400 characters.\n2. Emphasize 'UGC-style', 'handheld smartphone camera', 'subtle micro-shakes'.\n3. Describe natural movements (e.g., 'subtle head tilt', 'natural blinking', 'casually showing the product').\n4. Maintain a photorealistic and authentic vibe."
                        },
                        {
                            "type": "image_url", 
                            "image_url": {"url": f"data:image/jpeg;base64,{base64_image_for_gpt}"}
                        }
                    ]
                }
            ],
            max_tokens=150,
            temperature=0.7
        )
        return response.choices[0].message.content.strip()
    except Exception as e:
        print(f"GPT Error: {str(e)}")
        return "Natural handheld UGC-style video, subtle micro-shakes, smiling and showing product, highly realistic."


def create_photo(blogger_images, product_images):
    logs = ""
    if not blogger_images or not product_images:
        raise gr.Error("Please upload photos of both the blogger and the product!")
    
    logs += "[System] Initializing process...\n"
    yield logs, None
    time.sleep(1)
    
    if not HAS_GENAI:
        logs += f"[Fatal Error] Google GenAI library is missing! Add 'google-genai' to requirements.txt! ({GENAI_ERR})\n"
        yield logs, None
        return

    # DEMO FALLBACK
    if not GOOGLE_API_KEY:
        logs += "[Warning] GOOGLE_API_KEY is not set! Running in Demo Fallback mode...\n"
        yield logs, None
        time.sleep(2)
        logs += "[Success] Photo generated! Loading into interface...\n"
        yield logs, "https://images.unsplash.com/photo-1611162617474-5b21e879e113?q=80&w=1000&auto=format&fit=crop"
        return

    # REAL API CALL
    try:
        logs += "[Storage] Preparing blogger and product images...\n"
        yield logs, None
        
        client = genai.Client(api_key=GOOGLE_API_KEY)
        contents =[SYSTEM_PROMPT]
        
        for img in blogger_images:
            path = img if isinstance(img, str) else img.name
            contents.append(Image.open(path))
            
        for img in product_images:
            path = img if isinstance(img, str) else img.name
            contents.append(Image.open(path))
            
        logs += f"[Google GenAI] Sending API request (gemini-3-pro-image-preview)...\n"
        yield logs, None
        
        resp = client.models.generate_content(
            model="gemini-3-pro-image-preview",
            contents=contents,
            config=types.GenerateContentConfig(
                response_modalities=['TEXT', 'IMAGE'],
                image_config=types.ImageConfig(aspect_ratio="3:4", image_size="2K")
            )
        )
        
        img_data = None
        for part in resp.parts:
            if part.inline_data: 
                img_data = part.inline_data.data
                break
                
        if img_data:
            logs += "[Success] Photo generated successfully! Loading to UI...\n"
            result_img = Image.open(io.BytesIO(img_data))
            yield logs, result_img
        else:
            logs += "[Error] API succeeded but returned no image data.\n"
            yield logs, None
            
    except Exception as e:
        err = str(e)
        if "429" in err:
            logs += f"[Error] Rate limited by API (429)\n"
        else:
            logs += f"[Exception] API Error: {err[:100]}\n"
        yield logs, None


def create_video(generated_photo_path):
    logs = ""
    gpt_prompt = ""
    
    if not generated_photo_path:
        raise gr.Error("Please click 'Create Photo' first to generate an image for the video!")

    logs += "[System] Initializing Video Pipeline...\n"
    yield logs, gpt_prompt, None
    time.sleep(0.5)

    # --- 1. ПОДГОТОВКА ИЗОБРАЖЕНИЯ ---
    logs += "[Storage] Preparing base image for analysis and animation...\n"
    yield logs, gpt_prompt, None
    
    img_obj = Image.open(generated_photo_path)
    if img_obj.mode != 'RGB':
        img_obj = img_obj.convert('RGB')
        
    buffered = io.BytesIO()
    img_obj.save(buffered, format="JPEG", quality=95)
    base64_image_clean = base64.b64encode(buffered.getvalue()).decode('utf-8')
    
    # --- 2. ВЫЗОВ GPT-4o VISION ---
    logs += "[GPT-4o Vision] Analyzing generated photo to write perfect Kling prompt...\n"
    yield logs, gpt_prompt, None
    
    gpt_prompt = generate_video_prompt_with_gpt(base64_image_clean)
    
    logs += "[GPT-4o Vision] Prompt generated successfully! Sending to Kling AI...\n"
    yield logs, gpt_prompt, None

    # --- 3. ВЫЗОВ KLING API ---
    if not KLING_ACCESS_KEY or not KLING_SECRET_KEY:
        logs += "[Warning] Kling API Keys are not set! Running in Demo Fallback mode...\n"
        yield logs, gpt_prompt, None
        time.sleep(2)
        logs += "[Success] Status: SUCCEED. Video is ready!\n"
        yield logs, gpt_prompt, "https://www.w3schools.com/html/mov_bbb.mp4"
        return

    try:
        api_token = generate_kling_token(KLING_ACCESS_KEY, KLING_SECRET_KEY)
        
        headers = {
            "Authorization": f"Bearer {api_token}",
            "Content-Type": "application/json"
        }
        
        payload = {
            "model_name": "kling-v2-6",
            "image": base64_image_clean, 
            "prompt": gpt_prompt,
            "negative_prompt": "blur, distort, low quality, bad anatomy, artificial, overacted",
            "duration": "5",
            "mode": "pro",
            "sound": "off"
        }

        logs += f"[Kling API] POST /v1/videos/image2video -> Starting task...\n"
        yield logs, gpt_prompt, None
        
        post_url = "https://api-singapore.klingai.com/v1/videos/image2video"
        response = requests.post(post_url, headers=headers, json=payload)
        resp_json = response.json()
        
        if response.status_code != 200 or resp_json.get("code") != 0:
            error_msg = resp_json.get('message', 'Unknown error')
            logs += f"[Error] API Creation Failed: {error_msg}\n"
            yield logs, gpt_prompt, None
            return
            
        task_id = resp_json.get("data", {}).get("task_id")
        logs += f"[Kling API] Task ID: {task_id}. Status: SUBMITTED...\n"
        yield logs, gpt_prompt, None
        
        get_url = f"https://api-singapore.klingai.com/v1/videos/image2video/{task_id}"
        
        while True:
            time.sleep(5) 
            poll_resp = requests.get(get_url, headers=headers)
            poll_json = poll_resp.json()
            
            if poll_json.get("code") != 0:
                logs += f"[Error] API Polling Failed: {poll_json.get('message')}\n"
                yield logs, gpt_prompt, None
                break
                
            data = poll_json.get("data", {})
            task_status = data.get("task_status", "").lower()
            
            if task_status == "succeed":
                task_result = data.get("task_result", {})
                videos = task_result.get("videos",[])
                
                if videos and "url" in videos[0]:
                    video_url = videos[0]["url"]
                    logs += "[Success] Status: SUCCEED. Video generation finished!\n"
                    yield logs, gpt_prompt, video_url
                else:
                    logs += "[Error] Status SUCCEED, but no video URL found.\n"
                    yield logs, gpt_prompt, None
                break
                
            elif task_status == "failed":
                fail_msg = data.get("task_status_msg", "Unknown reason")
                logs += f"[Error] Task FAILED. Reason: {fail_msg}\n"
                yield logs, gpt_prompt, None
                break
                
            else:
                logs += f"[Kling API] Status: {task_status.upper()}... Waiting 5s...\n"
                yield logs, gpt_prompt, None
                
    except Exception as e:
        logs += f"[Exception] Internal error occurred: {str(e)}\n"
        yield logs, gpt_prompt, None


# --- CUSTOM UI STYLES ---
# Я УБРАЛ color: #374151 !important; ИЗ .prompt-box, ТЕПЕРЬ ТЕКСТ БУДЕТ ВИДНО И В ТЕМНОЙ, И В СВЕТЛОЙ ТЕМЕ!
custom_css = (
    ".container { max-width: 1200px; margin: auto; }\n"
    ".output-media { border-radius: 16px; box-shadow: 0 10px 25px rgba(0,0,0,0.1); }\n"
    "button.primary { background: linear-gradient(90deg, #6366f1, #a855f7); border: none; }\n"
    "button.primary:hover { transform: translateY(-2px); box-shadow: 0 5px 15px rgba(168, 85, 247, 0.4); transition: all 0.3s ease; }\n"
    ".log-window textarea { background-color: #1e1e1e !important; color: #4ade80 !important; font-family: 'Courier New', Courier, monospace !important; font-size: 13px !important; }\n"
    ".prompt-box textarea { font-size: 14px !important; }" 
)

with gr.Blocks() as demo:
    gr.Markdown("<h1 style='text-align: center;'>✨ AI UGC Content Pipeline</h1>")
    gr.Markdown("<p style='text-align: center; color: gray;'>GenAI (Photo) ➡️ GPT-4o Vision (Analysis) ➡️ Kling AI (Video)</p>")
    
    with gr.Row():
        with gr.Column(scale=1):
            gr.Markdown("### 📥 1. Image Inputs")
            with gr.Group():
                blogger_imgs = gr.File(label="📸 Blogger (1-5 photos)", file_count="multiple", file_types=["image"])
                product_imgs = gr.File(label="🛍️ Product (1-5 photos)", file_count="multiple", file_types=["image"])
            
            gr.Markdown("### 🧠 2. AI Prompts")
            banana_prompt_ui = gr.Textbox(
                label="GenAI Photo Prompt (Static)", 
                value=SYSTEM_PROMPT, 
                interactive=False, 
                lines=4,
                elem_classes="prompt-box"
            )
            
            kling_prompt_ui = gr.Textbox(
                label="Kling Video Prompt (Auto-generated by GPT-4o Vision)", 
                placeholder="Click 'Create Video'. GPT-4o will analyze the generated photo and write the perfect prompt here...", 
                interactive=True, 
                lines=4,
                elem_classes="prompt-box"
            )

            with gr.Row():
                btn_photo = gr.Button("Step 1: Create Photo", variant="primary", size="lg")
                btn_video = gr.Button("Step 2: Create Video", variant="primary", size="lg")
                
            logs_output = gr.Textbox(
                label="🖥️ System Logs", 
                lines=6, 
                interactive=False, 
                elem_classes="log-window",
                placeholder="Generation logs will appear here in real-time..."
            )
                
        with gr.Column(scale=1):
            gr.Markdown("### 📤 3. Outputs")
            
            with gr.Tabs():
                with gr.Tab("Photo Result"):
                    out_photo = gr.Image(label="Generated Photo", elem_classes="output-media", type="filepath")
                with gr.Tab("Video Result"):
                    out_video = gr.Video(label="Animated UGC Video", elem_classes="output-media")
            
    btn_photo.click(
        fn=create_photo, 
        inputs=[blogger_imgs, product_imgs], 
        outputs=[logs_output, out_photo]
    )
    
    btn_video.click(
        fn=create_video, 
        inputs=[out_photo], 
        outputs=[logs_output, kling_prompt_ui, out_video]
    )

if __name__ == "__main__":
    demo.launch(
        theme=gr.themes.Soft(primary_hue="indigo", secondary_hue="violet"), 
        css=custom_css
    )