File size: 38,199 Bytes
7aaa385
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
# -*- coding: utf-8 -*-
import os
import re
import io
import json
import base64
import hashlib
import sqlite3
import urllib.request
import urllib.error

import numpy as np
import folder_paths
import comfy.sd
import comfy.lora
import comfy.utils
from safetensors import safe_open
from comfy_execution.graph_utils import ExecutionBlocker

try:
    from PIL import Image
    _HAS_PIL = True
except Exception:
    _HAS_PIL = False

CACHE_FILE = os.path.join(os.path.dirname(__file__), "dolphin_tags_cache.json")


# ======================================================================
#  λͺ¨λΈ 프리셋 (ν…μŠ€νŠΈ LLM = 4μƒ· λΆ„λ°° / λΉ„μ „ VLM = 이미지 인식)
# ======================================================================
LLM_PRESETS = {
    "πŸ”ž DeepSeek V3.2 (λ¬΄μ‚­μ œΒ·JSONμ•ˆμ •Β·μΆ”μ²œ)": "deepseek/deepseek-v3.2",
    "πŸ”ž Magnum v4 72B (ν‘œν˜„ μžμ—°μŠ€λŸ¬μ›€)": "anthracite-org/magnum-v4-72b",
    "πŸ”ž Cydonia 24B v4.1 (κ²½λŸ‰Β·λΉ λ¦„)": "thedrummer/cydonia-24b-v4.1",
    "πŸ”ž Dolphin Mistral Venice (무료)": "cognitivecomputations/dolphin-mistral-24b-venice-edition:free",
    "βš™οΈ custom (μ•„λž˜ ν…μŠ€νŠΈ μ‚¬μš©)": "__custom__",
}
LLM_PRESET_LABELS = list(LLM_PRESETS.keys())

# λΉ„μ „ λͺ¨λΈ: 특수 상황 이해 + λ‹€κ΅­μ–΄. Qwen2.5-VL 계열이 포즈/λ§₯락 νŒŒμ•…μ— 강함.
VLM_PRESETS = {
    "πŸ‘ Qwen2.5-VL 72B (νŠΉμˆ˜μƒν™©Β·μΆ”μ²œ)": "qwen/qwen2.5-vl-72b-instruct",
    "πŸ‘ Qwen2.5-VL 32B (κ· ν˜•)": "qwen/qwen2.5-vl-32b-instruct",
    "πŸ‘ Qwen2.5-VL 32B (무료)": "qwen/qwen2.5-vl-32b-instruct:free",
    "πŸ‘ Mistral Small 3.1 24B (Pixtral)": "mistralai/mistral-small-3.1-24b-instruct",
    "πŸ‘ Gemma 3 27B (κ²½λŸ‰)": "google/gemma-3-27b-it",
    "βš™οΈ custom (μ•„λž˜ ν…μŠ€νŠΈ μ‚¬μš©)": "__custom__",
}
VLM_PRESET_LABELS = list(VLM_PRESETS.keys())


def resolve_slug(table, label, custom_text, default):
    slug = table.get(label, "__custom__")
    if slug == "__custom__":
        return custom_text.strip() or default
    return slug


def resolve_api_key(widget_value):
    """μœ„μ ―μ— ν‚€κ°€ 있으면 κ·ΈλŒ€λ‘œ μ‚¬μš©. λΉ„μ–΄μžˆμœΌλ©΄ OPENROUTER_API_KEY ν™˜κ²½λ³€μˆ˜λ‘œ 폴백.
    -> μ›Œν¬ν”Œλ‘œμš° JSON(곡유/λ°±μ—… μ‹œ 유좜 μœ„ν—˜)에 μ‹€ν‚€λ₯Ό λ°•μ•„λ‘˜ ν•„μš”κ°€ 없어짐."""
    key = (widget_value or "").strip()
    if key:
        return key
    return os.environ.get("OPENROUTER_API_KEY", "").strip()


# ======================================================================
#  μΊμ‹œ & 트리거 (κ²€μ¦λœ 둜직 μž¬μ‚¬μš©)
# ======================================================================
def _load_cache():
    if os.path.exists(CACHE_FILE):
        try:
            with open(CACHE_FILE, "r", encoding="utf-8") as f:
                return json.load(f)
        except (json.JSONDecodeError, OSError):
            return {}
    return {}


def _save_cache(cache):
    try:
        with open(CACHE_FILE, "w", encoding="utf-8") as f:
            json.dump(cache, f, indent=4, ensure_ascii=False)
    except OSError:
        pass


def _file_sig(path):
    st = os.stat(path)
    return f"{os.path.basename(path)}::{st.st_size}::{int(st.st_mtime)}"


def _sha256(path):
    h = hashlib.sha256()
    with open(path, "rb") as f:
        for block in iter(lambda: f.read(4096 * 1024), b""):
            h.update(block)
    return h.hexdigest()


# CivitAI 및 미러(civitaired) ν•΄μ‹œ 쑰회 μ—”λ“œν¬μΈνŠΈ. μ•žμ—μ„œλΆ€ν„° μˆœμ„œλŒ€λ‘œ μ‹œλ„.
TRIGGER_HASH_APIS = [
    "https://civitai.com/api/v1/model-versions/by-hash/{h}",
    "https://civitaired.com/api/v1/model-versions/by-hash/{h}",
]

# LoRA Manager(comfyui-lora-manager)κ°€ μŠ€μΊ” μ‹œ λ§Œλ“€μ–΄λ‘λŠ” 둜컬 SQLite μΊμ‹œ.
# sha256으둜 trained_wordsλ₯Ό μ¦‰μ‹œ 쑰회 κ°€λŠ₯ -> λ„€νŠΈμ›Œν¬ 호좜 없이 νŠΈλ¦¬κ±°μ›Œλ“œ 확보.
LORA_MANAGER_DB = os.path.join(
    os.environ.get("LOCALAPPDATA", ""), "ComfyUI-LoRA-Manager", "cache", "model", "comfyui.sqlite")


def _query_lora_manager_db(digest):
    """LoRA Manager SQLite μΊμ‹œμ—μ„œ sha256으둜 trained_words 쑰회. μ‹€νŒ¨/미발견 μ‹œ None."""
    if not digest or not os.path.exists(LORA_MANAGER_DB):
        return None
    try:
        uri = f"file:{LORA_MANAGER_DB}?mode=ro"
        con = sqlite3.connect(uri, uri=True, timeout=3)
        try:
            cur = con.cursor()
            cur.execute(
                "SELECT trained_words FROM models WHERE model_type='lora' AND sha256=? LIMIT 1",
                (digest,))
            row = cur.fetchone()
        finally:
            con.close()
        if not row or not row[0]:
            return None
        tw = json.loads(row[0])
        return tw if isinstance(tw, list) else None
    except (sqlite3.Error, OSError, ValueError, json.JSONDecodeError) as e:
        print(f"⚠️ [Director] LoRA Manager DB 쑰회 μ‹€νŒ¨: {type(e).__name__}: {e}")
        return None


def _query_trigger_api(url):
    """단일 ν•΄μ‹œ 쑰회 μ—”λ“œν¬μΈνŠΈμ—μ„œ trainedWords 리슀트λ₯Ό λ°˜ν™˜. μ‹€νŒ¨ μ‹œ ([], μ‚¬μœ )."""
    try:
        req = urllib.request.Request(url, headers={'User-Agent': 'Mozilla/5.0'})
        with urllib.request.urlopen(req, timeout=8) as r:
            data = json.loads(r.read().decode('utf-8'))
            return (data.get("trainedWords", []) or []), None
    except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError,
            OSError, ValueError, json.JSONDecodeError) as e:
        return [], f"{type(e).__name__}: {e}"


def get_triggers(lora_name, cache):
    if lora_name == "None":
        return []
    path = folder_paths.get_full_path("loras", lora_name)
    if not path or not os.path.exists(path):
        return []
    try:
        with safe_open(path, framework="pt", device="cpu") as f:
            m = f.metadata()
            if m and "modelspec.trigger_words" in m:
                tw = json.loads(m["modelspec.trigger_words"])
                if isinstance(tw, list):
                    return tw
                if isinstance(tw, str):
                    return [tw]
    except (OSError, ValueError, json.JSONDecodeError):
        pass
    try:
        sig = _file_sig(path)
    except OSError:
        sig = lora_name
    # μΊμ‹œμ— 'λΉ„μ–΄μžˆμ§€ μ•Šμ€' κ²°κ³Όκ°€ μžˆμ„ λ•Œλ§Œ μ‹ λ’°. 과거에 μ‹€νŒ¨λ‘œ μ €μž₯된 빈 값은
    # λ¬΄μ‹œν•˜κ³  μž¬μ‘°νšŒν•œλ‹€ (빈 κ°’ μ˜κ΅¬ν™” νšŒκ·€ λ°©μ§€).
    cached = cache.get(sig)
    if cached:
        return cached

    # ν•΄μ‹œλŠ” 1회만 κ³„μ‚°ν•΄μ„œ LoRA Manager DB -> CivitAI -> civitaired 순으둜 쑰회.
    result, base = [], os.path.basename(path)
    try:
        digest = _sha256(path)
    except OSError as e:
        print(f"⚠️ [Director] ν•΄μ‹œ 계산 μ‹€νŒ¨({base}): {e}")
        digest = None
    if digest:
        found = _query_lora_manager_db(digest)
        if found:
            print(f"βœ… [Director] 트리거 쑰회 성곡({base}) @ LoRA Manager DB: {found}")
            result = found
    if digest and not result:
        for tmpl in TRIGGER_HASH_APIS:
            host = tmpl.split('/')[2]
            found, err = _query_trigger_api(tmpl.format(h=digest))
            if found:
                print(f"βœ… [Director] 트리거 쑰회 성곡({base}) @ {host}: {found}")
                result = found
                break
            print(f"⚠️ [Director] 트리거 쑰회 μ‹€νŒ¨({base}) @ {host}: {err or 'no trainedWords'}")

    # 성곡(λΉ„μ–΄μžˆμ§€ μ•ŠμŒ)일 λ•Œλ§Œ 캐싱. 빈 κ²°κ³ΌλŠ” μ €μž₯ν•˜μ§€ μ•Šμ•„ λ‹€μŒ μ‹€ν–‰μ—μ„œ μž¬μ‹œλ„λ¨.
    if result:
        cache[sig] = result
        _save_cache(cache)
    return result


def apply_model_lora(model, lora_data, weight):
    try:
        return comfy.sd.load_lora_for_models(model, None, lora_data, weight, 0)[0]
    except (AttributeError, TypeError):
        new_model = model.clone()
        key_map = comfy.lora.model_lora_keys_unet(new_model.model)
        loaded = comfy.lora.load_lora(lora_data, key_map)
        new_model.add_patches(loaded, weight)
        return new_model


def tensor_to_base64(image_tensor, max_side=1024):
    """ComfyUI IMAGE ν…μ„œ(B,H,W,C, 0~1 float) 첫 ν”„λ ˆμž„μ„ JPEG base64 data URI둜."""
    if not _HAS_PIL:
        print("❌ [Director] PIL(Pillow) μ—†μŒ - λΉ„μ „ λΆˆκ°€. `pip install Pillow` ν•„μš”.")
        return None
    if image_tensor is None:
        print("❌ [Director] image μž…λ ₯이 None - IMAGE μ—°κ²° 확인.")
        return None
    try:
        img = image_tensor
        if hasattr(img, "cpu"):
            img = img.cpu().numpy()
        img = np.asarray(img)
        if img.ndim == 4:
            img = img[0]
        arr = np.clip(img * 255.0, 0, 255).astype(np.uint8)
        pil = Image.fromarray(arr)
        w, h = pil.size
        if max(w, h) > max_side:
            scale = max_side / float(max(w, h))
            pil = pil.resize((int(w * scale), int(h * scale)), Image.LANCZOS)
        buf = io.BytesIO()
        pil.convert("RGB").save(buf, format="JPEG", quality=90)
        b64 = base64.b64encode(buf.getvalue()).decode("utf-8")
        print(f"πŸ–Ό [Director] image encoded ok: {w}x{h} -> {pil.size}, {len(b64)//1024}KB base64")
        return f"data:image/jpeg;base64,{b64}"
    except Exception as e:
        print(f"❌ [Director] image encode failed: {type(e).__name__}: {e}")
        return None


# ======================================================================
#  Vision Director: 이미지 인식 + ν•„μˆ˜λ™μž‘ μ‹€ν–‰ + 클립당 4둜라
# ======================================================================
class DolphinVisionDirector:
    """
    이미지 ν•˜λ‚˜μ™€ 'ν•„μˆ˜ λ™μž‘'κ³Ό 둜라만 λ„£μœΌλ©΄ 4개 연속 클립이 μ™„μ„±λ˜λŠ” μ˜¬μΈμ› λ…Έλ“œ.

      1) IMAGE -> λΉ„μ „ VLM 이 상황(캐릭터 μƒνƒœ/μžμ„Έ/ν™˜κ²½)만 μ•΅μ»€λ‘œ νŒŒμ•…
      2) κ·Έ λ§₯락 μœ„μ—μ„œ 'λ‚΄κ°€ μ€€ ν•„μˆ˜ λ™μž‘(essential_actions)'만 4μƒ·μœΌλ‘œ λΆ„λ°°
      3) ν΄λ¦½λ§ˆλ‹€ 둜라 4κ°œμ”© κ°œλ³„ 적용(μ›¨μ΄νŠΈλ„ κ°œλ³„) + 트리거 μžλ™ μ£Όμž…
    """

    @classmethod
    def INPUT_TYPES(s):
        loras = ["None"] + (folder_paths.get_filename_list("loras") or [])
        req = {
            "image": ("IMAGE",),
            "model_base": ("MODEL",),
            "model_high": ("MODEL",),
            "model_low": ("MODEL",),
            "essential_actions": ("STRING", {"multiline": True,
                "default": ("λŒμ§„ν•œλ‹€, 검을 νœ˜λ‘˜λŸ¬ 적을 μ“°λŸ¬λœ¨λ¦°λ‹€, μ΄μ•Œμ„ νŠ•κ²¨λ‚Έλ‹€, μˆ¨ν†΅μ„ λŠλŠ”λ‹€")}),
            "num_clips": (["4 (20s)", "3 (15s)", "2 (10s)"], {"default": "4 (20s)",
                "tooltip": "μ‹€μ œλ‘œ λ Œλ”λ§ν•  클립 수. Extend range(Fast Groups Bypasser) ν† κΈ€κ³Ό λ§žμΆ°μ„œ "
                           "μ„€μ •ν•΄μ•Ό μŠ€ν† λ¦¬κ°€ κ·Έ 길이에 맞게 배뢄됨. λ‚˜λ¨Έμ§€ 클립 μŠ¬λ‘―μ€ λ§ˆμ§€λ§‰ 클립 λ‚΄μš©μ„ 반볡."}),
            "character_name": ("STRING", {"default": "AUTO"}),
            "artistic_vibe": ("STRING", {"multiline": True,
                                         "default": "cinematic lighting, dynamic motion"}),
            "pacing": (["auto", "slow-burn (1 beat)", "balanced (multi beat)", "rapid (4 distinct)"],
                       {"default": "auto"}),
            "detail_level": (["짧게(κ°„κ²°)", "ν•„μˆ˜+μ•΅μ»€λ§Œ", "상세"], {"default": "ν•„μˆ˜+μ•΅μ»€λ§Œ"}),
            "inject_triggers": ("BOOLEAN", {"default": True,
                                            "label_on": "TRIGGERS ON", "label_off": "TRIGGERS OFF"}),
            "use_vision": ("BOOLEAN", {"default": True,
                                       "label_on": "VISION ON", "label_off": "VISION OFF (text only)"}),
            "openrouter_api_key": ("STRING", {"default": ""}),
            "vlm_preset": (VLM_PRESET_LABELS, {"default": "πŸ‘ Qwen2.5-VL 72B (νŠΉμˆ˜μƒν™©Β·μΆ”μ²œ)"}),
            "vlm_model": ("STRING", {"default": "qwen/qwen2.5-vl-72b-instruct"}),
            "llm_preset": (LLM_PRESET_LABELS, {"default": "πŸ”ž DeepSeek V3.2 (λ¬΄μ‚­μ œΒ·JSONμ•ˆμ •Β·μΆ”μ²œ)"}),
            "llm_model": ("STRING", {"default": "deepseek/deepseek-v3.2"}),
            "creativity": ("FLOAT", {"default": 0.7, "min": 0.1, "max": 1.5, "step": 0.05}),
            "seed": ("INT", {"default": 0, "min": 0, "max": 0xffffffffffffffff}),
        }
        opt = {}
        for c in range(1, 5):
            for n in range(1, 5):
                opt[f"c{c}_lora_{n}"] = (loras, {"default": "None"})
                opt[f"c{c}_w_{n}"] = ("STRING", {"default": "1.0, 1.0, 1.0", "multiline": False})
        opt["manual_triggers"] = ("STRING", {"default": "", "multiline": True})
        opt["vision_caption"] = ("STRING", {"multiline": True, "default": "",
                                            "forceInput": True})
        return {"required": req, "optional": opt}

    RETURN_TYPES = (
        "MODEL", "MODEL", "MODEL", "STRING",
        "MODEL", "MODEL", "MODEL", "STRING",
        "MODEL", "MODEL", "MODEL", "STRING",
        "MODEL", "MODEL", "MODEL", "STRING",
        "STRING", "STRING", "STRING",
        "BOOLEAN", "BOOLEAN", "BOOLEAN", "BOOLEAN",
    )
    RETURN_NAMES = (
        "c1_base", "c1_high", "c1_low", "c1_prompt",
        "c2_base", "c2_high", "c2_low", "c2_prompt",
        "c3_base", "c3_high", "c3_low", "c3_prompt",
        "c4_base", "c4_high", "c4_low", "c4_prompt",
        "triggers", "scene_analysis", "debug_info",
        "c1_is_final", "c2_is_final", "c3_is_final", "c4_is_final",
    )
    FUNCTION = "direct"
    CATEGORY = "Dolphin"

    def _build_final(self, action, master_scene, name, trigger_str):
        head = ", ".join([p for p in [name, trigger_str] if p])
        body = " ".join([p for p in [master_scene.strip().strip(",. "), action.strip()] if p])
        if head and body:
            return f"{head}\n{body}"
        return head or body

    def _pacing_hint(self, p):
        return {
            "auto": "Decide pacing yourself based on how many distinct beats the actions contain.",
            "slow-burn (1 beat)": "One climactic beat: clips 1-3 build tension, clip 4 detonates.",
            "balanced (multi beat)": "Early clips build, later clips deliver; keep motion continuous.",
            "rapid (4 distinct)": "Four distinct consecutive actions, one per clip, each kinetic.",
        }.get(p, "Decide pacing yourself.")

    def _detail_hint(self, d):
        return {
            "짧게(κ°„κ²°)": "Keep each clip to a short punchy motion phrase. Pure movement verbs only.",
            "ν•„μˆ˜+μ•΅μ»€λ§Œ": "Each clip is one concise continuous motion. No description, only movement.",
            "상세": "Motion may be described in more detail, but ONLY body movement β€” no scenery.",
        }.get(d, "Motion only, no description.")

    def _extract_json(self, text):
        if not text:
            return None
        text = re.sub(r'```(?:json)?', '', text).strip()
        a, b = text.find('{'), text.rfind('}')
        if a == -1 or b <= a:
            return None
        blob = text[a:b + 1]
        try:
            return json.loads(blob)
        except json.JSONDecodeError:
            try:
                return json.loads(re.sub(r',\s*([}\]])', r'\1', blob))
            except json.JSONDecodeError:
                return None

    def _call_or(self, messages, model, key, temp, timeout=120, force_json=False):
        payload = {"model": model.strip(), "messages": messages, "temperature": temp}
        if force_json:
            payload["response_format"] = {"type": "json_object"}
        req = urllib.request.Request(
            "https://openrouter.ai/api/v1/chat/completions",
            data=json.dumps(payload).encode('utf-8'),
            headers={'Authorization': f'Bearer {key}', 'Content-Type': 'application/json'})
        raw = json.loads(urllib.request.urlopen(req, timeout=timeout).read().decode('utf-8'))
        if "choices" not in raw:
            err = raw.get("error") or raw
            msg = err.get("message") if isinstance(err, dict) else str(err)
            code = err.get("code") if isinstance(err, dict) else "?"
            raise ValueError(f"API error (code={code}): {msg}")
        return raw['choices'][0]['message']['content'].strip()

    def _vision_analyze(self, data_uri, vlm_model, key, temp):
        sys = ("You are a visual scene analyst for a video pipeline. "
               "Look at the image and extract ONLY concrete anchors, no storytelling. "
               "Report exactly these lines:\n"
               "(1) character appearance & clothing.\n"
               "(2) current pose/posture (sitting, crouched, standing, balance, limb positions).\n"
               "(3) SUPPORT & CONTACT: what the character is on top of, sitting on, holding, "
               "leaning against, or touching (e.g. 'seated balanced on a yoga ball', "
               "'gripping a sword', 'kneeling on the floor'). This is critical for physics β€” "
               "state it explicitly even if obvious.\n"
               "(4) environment & lighting (brief).\n"
               "Be terse, one short line each. English only. "
               "Do not invent actions or narrative. If content is mature, describe it plainly and factually.")
        messages = [
            {"role": "system", "content": sys},
            {"role": "user", "content": [
                {"type": "text", "text": "Extract the anchors from this image."},
                {"type": "image_url", "image_url": {"url": data_uri}},
            ]},
        ]
        import time
        last_err = "vision error"
        for attempt in range(1, 4):
            try:
                resp = self._call_or(messages, vlm_model, key, temp, timeout=90).strip()
                if not resp:
                    print(f"⚠️ [Director] VLM({vlm_model}) 응닡이 λΉ„μ—ˆμŒ (κ±°λΆ€ or 빈 λ°˜ν™˜)")
                    return "", "vision empty response"
                print(f"πŸ‘ [Director] VLM({vlm_model}) anchors (attempt {attempt}):\n{resp[:300]}")
                return resp, None
            except urllib.error.HTTPError as e:
                body = ""
                try:
                    body = e.read().decode('utf-8')[:200]
                except Exception:
                    pass
                print(f"❌ [Director] VLM HTTP {e.code} (attempt {attempt}): {body}")
                last_err = f"vision HTTP {e.code}"
                if e.code in (429, 503) and attempt < 3:
                    wait = 2 ** attempt
                    print(f"⏳ [Director] rate limited, {wait}s ν›„ μž¬μ‹œλ„...")
                    time.sleep(wait)
                    continue
                return "", last_err
            except (urllib.error.URLError, TimeoutError, KeyError, ValueError) as e:
                print(f"❌ [Director] VLM error (attempt {attempt}): {type(e).__name__}: {e}")
                last_err = f"vision error: {e}"
                if attempt < 3:
                    time.sleep(2 ** attempt)
                    continue
                return "", last_err
        return "", last_err

    def _fallback_name(self, anchors):
        import random
        a = (anchors or "").lower()
        east = ["asian", "east asian", "japanese", "korean", "chinese", "anime",
                "kimono", "hanbok", "qipao", "oriental"]
        western = ["western", "european", "caucasian", "american", "blonde", "redhead"]
        east_names = ["Yuna", "Kaede", "Jin", "Mei", "Haruka", "Rin", "Sora", "Aoi"]
        west_names = ["Elena", "Ryan", "Ava", "Lucas", "Mila", "Ethan", "Nora", "Leo"]
        if any(k in a for k in east):
            pool = east_names
        elif any(k in a for k in western):
            pool = west_names
        else:
            pool = east_names + west_names
        return random.choice(pool)

    def _split_actions_to_clips(self, actions, num_clips=4):
        parts = re.split(r'(?<=[.!?。])\s+|[,、]|\n+', actions)
        parts = [p.strip() for p in parts if len(p.strip()) > 1]
        if not parts:
            parts = [actions.strip()]
        n = len(parts)
        last = num_clips
        assign = {c: [] for c in range(1, num_clips + 1)}
        if num_clips == 1:
            assign[1] = parts
            return assign
        build_labels = ["(build-up toward)", "(begin)", "(escalate)", "(continue escalating)"]
        if n == 1:
            for c in range(1, last):
                label = build_labels[min(c - 1, len(build_labels) - 1)]
                assign[c] = [f"{label} {parts[0]}"]
            assign[last] = [parts[0]]
        elif n <= num_clips:
            assign[last] = [parts[-1]]
            head = parts[:-1]
            for i, ph in enumerate(head):
                assign[min(i + 1, last - 1)].append(ph)
            for c in range(1, last):
                if not assign[c]:
                    nxt = parts[min(c, n - 1)]
                    assign[c] = [f"(move toward) {nxt}"]
        else:
            assign[last] = [parts[-1]]
            rest = parts[:-1]
            k, m = divmod(len(rest), last - 1)
            idx = 0
            for c in range(1, last):
                end = idx + k + (1 if (c - 1) < m else 0)
                assign[c] = rest[idx:end]
                idx = end
        return assign

    def _plan_shots(self, anchors, actions, name, vibe, pacing, detail, llm_model, key, temp, seed, num_clips=4):
        assign = self._split_actions_to_clips(actions, num_clips)
        assign_block = "\n".join(
            f"  clip_{c}: {', '.join(assign[c]) if assign[c] else '(continue previous motion)'}"
            for c in range(1, num_clips + 1))

        last_clip = num_clips
        total_seconds = num_clips * 5
        if num_clips >= 3:
            chain = " and ".join(f"clip_{c}<-clip_{c-1}" for c in range(3, num_clips + 1))
            continuity_middle = (f". The same applies {chain}: each clip picks up exactly where the "
                                  f"previous clip's body position left off")
        else:
            continuity_middle = ""
            
        continuity_rule = ""
        if num_clips >= 2:
            continuity_rule = (
                "3) INTER-CLIP CONTINUITY. clip_2 MUST begin from the exact physical end-state of "
                "clip_1's action (same position, body orientation, and momentum) β€” it continues the "
                "motion, it does not restart from a neutral stance" + continuity_middle +
                f", as one unbroken {total_seconds}-second performance rather than {num_clips} separate poses. Do "
                "NOT explicitly narrate the hand-off (no phrases like 'continuing from', 'still', 'as "
                "before', 'picking up where') β€” simply choreograph the new action so it is kinematically "
                "consistent with how the previous action would have left the body positioned. This is a "
                "physical-continuity constraint, not a scripted transition line.\n"
            )

        json_clip_fields = ",".join(
            f'"clip_{i}":"<only clip_{i} action{", the finisher" if i == last_clip else ""}, '
            f'continuous & fast>"' for i in range(1, num_clips + 1))

        sys = (
            f"You are an elite fight/action choreographer for a {total_seconds}-second video made of "
            f"{num_clips} consecutive 5-second clip{'s' if num_clips != 1 else ''}. You are given SCENE ANCHORS "
            "(the character's starting look/pose/place from an image) and a STRICT per-clip action "
            "assignment.\n"
            "\n"
            "ABSOLUTE RULES:\n"
            "1) FOLLOW THE ASSIGNMENT EXACTLY. Each clip animates ONLY the action(s) assigned "
            "to it below. You MUST NOT move a later clip's action into an earlier clip. "
            f"clip_1 must NOT contain clip_{last_clip}'s action. The final action happens ONLY in "
            f"clip_{last_clip}. "
            "This is the most important rule - violating the order is a failure. "
            "(ONE exception: clip_1 may prepend a brief transition OUT of the starting physics from "
            "rule 2 β€” e.g. leaving the yoga ball β€” before its assigned action. This is not borrowing "
            "a later action; it is grounding the first one.)\n"
            "2) STARTING PHYSICS (use the anchors here). Read the character's CURRENT pose AND "
            "what they are supported by / in contact with from the SCENE ANCHORS (e.g. seated on a "
            "yoga ball, kneeling, gripping a weapon, leaning on a wall). clip_1 MUST begin the first "
            "action FROM that exact physical situation and honor its constraints. If the support is "
            "unstable or unusual (a yoga ball, a ledge, a moving surface), the motion MUST account "
            "for it β€” e.g. anchor 'seated balanced on a yoga ball' -> clip_1 = 'pushing off the "
            "wobbling ball and rising to their feet as it rolls away' BEFORE any further action. "
            "Never ignore or contradict the starting support (do not stand if seated on the ball "
            "without first leaving it; do not assume a weapon is drawn if the anchor shows it "
            "sheathed). This makes the motion physically continuous with the input image. Reference "
            "the support ONLY to ground how the motion STARTS β€” do NOT describe its appearance, "
            "color, or the wider environment.\n"
            + continuity_rule +
            "4) ACTION ONLY. Write physical body movement. NO scenery, NO lighting, NO clothing, "
            "NO mood, NO camera talk. If a word is not a movement or body part, delete it. "
            "(Referring to the starting pose in clip_1 per rule 2, or to the prior clip's ending "
            "position per rule 3, is allowed since it describes how the body moves.)\n"
            "5) FILL 5 SECONDS. Expand the assigned action into a single continuous flow of motion "
            "that occupies the whole clip (use 'then', 'immediately', 'without pausing'). "
            "Do not borrow the next action to fill time - elaborate the CURRENT action instead.\n"
            "6) ANTI-SLOW-MOTION: real-time speed, kinetic adverbs (swiftly, rapidly, explosively, "
            "in a split second). NO slow motion, NO freeze, NO holding a pose.\n"
            "7) CHARACTER NAME: study the SCENE ANCHORS. Judge the character's apparent ethnicity/"
            "setting (East Asian, Western, etc.) and INVENT a specific fitting first name that "
            "matches it (e.g. East Asian -> 'Yuna','Kaede','Jin'; Western -> 'Elena','Ryan'). "
            "NEVER output 'AUTO' or an empty name. If a NAME is given below (not 'AUTO'), use it.\n"
            f"8) {detail}\n"
            "9) MOTION ENERGY: a MOTION STYLE cue is given below (e.g. 'cinematic lighting, dynamic "
            "motion'). Use it ONLY to calibrate how forceful/graceful/frantic the movement FEELS "
            "(word choice, verb intensity). Do NOT quote it and do NOT let any scenery/lighting/mood "
            "words from it leak into the output β€” rule 4 still applies.\n"
            "\n"
            "master_scene: leave it EMPTY.\n"
            "\n"
            'OUTPUT ONLY JSON: {"character":"<a real name, never AUTO>","master_scene":"",'
            + json_clip_fields + "}"
        )
        usr = (f"NAME: {name}\nPACING: {self._pacing_hint(pacing)}\n\n"
               f"MOTION STYLE (kinetic energy cue only, per rule 9 β€” do not describe scenery/"
               f"lighting/mood from this): {vibe.strip() or '(none)'}\n\n"
               f"SCENE ANCHORS (use the POSE to start clip_1's motion per rule 2; "
               f"use overall look only for naming β€” do NOT copy look into the clips):\n"
               f"{anchors or '(none β€” assume a neutral ready stance)'}\n\n"
               f"STRICT PER-CLIP ACTION ASSIGNMENT (animate ONLY what is listed per clip, "
               f"in this order, finisher in clip_{last_clip}):\n{assign_block}\n")
        messages = [{"role": "system", "content": sys}, {"role": "user", "content": usr}]
        t = temp
        last = "⚠️ plan failed (no attempts ran)"
        for attempt in (1, 2):
            try:
                content = self._call_or(messages, llm_model, key, t, force_json=True)
                parsed = self._extract_json(content)
                if parsed and all(parsed.get(f"clip_{i}") for i in range(1, num_clips + 1)):
                    return parsed, f"βœ… plan ok (attempt {attempt})"
                last = f"⚠️ incomplete JSON (attempt {attempt}): {content[:150]!r}"
                print(f"⚠️ [Director] plan attempt {attempt} returned incomplete JSON: {content[:300]}")
            except urllib.error.HTTPError as e:
                body = ""
                try:
                    body = e.read().decode('utf-8')[:200]
                except Exception:
                    pass
                last = f"❌ plan HTTP {e.code} (attempt {attempt}): {body}"
                print(f"❌ [Director] plan {last}")
            except (urllib.error.URLError, TimeoutError, KeyError, ValueError) as e:
                last = f"❌ plan error (attempt {attempt}): {type(e).__name__}: {e}"
                print(f"❌ [Director] {last}")
            t = min(temp, 0.4)
        return None, last

    def _local_split(self, actions, num_clips=4):
        parts = re.split(r'(?<=[.!?。])\s+|[,、]|\n+', actions)
        parts = [p.strip() for p in parts if len(p.strip()) > 1]
        if not parts:
            parts = [actions.strip()]
        n = len(parts)
        last = num_clips - 1

        def fast(phrase, lead="swiftly"):
            p = phrase.strip()
            return f"{lead} {p}, continuous real-time motion, no slow motion" if p else ""

        if num_clips == 1:
            return [fast(", ".join(parts), "explosively")]

        if n == 1:
            a = parts[0]
            templates = [
                f"rapidly moves into position toward {a}, no pause",
                f"immediately begins {a}, brisk continuous motion, no slow motion",
                f"presses {a} without stopping, fast decisive movement",
            ]
            out = [templates[min(i, len(templates) - 1)] for i in range(num_clips - 1)]
            out.append(f"explosively completes {a} at real-time speed, no slow motion, no freeze")
        elif n <= num_clips:
            out = [""] * num_clips
            out[last] = fast(parts[-1], "explosively")
            head = parts[:-1]
            for i, ph in enumerate(head):
                s = min(i, num_clips - 2)
                out[s] = (out[s] + ", then " + fast(ph)).strip(", ") if out[s] else fast(ph)
            for i in range(num_clips - 1):
                if not out[i]:
                    key = parts[min(i, n - 1)]
                    out[i] = f"rapidly moves toward {key}, brisk continuous motion, no slow motion"
        else:
            k, m = divmod(n - 1, num_clips - 1)
            idx, chunks = 0, []
            for i in range(num_clips - 1):
                end = idx + k + (1 if i < m else 0)
                seg = parts[idx:end]
                idx = end
                chunks.append(", then ".join(fast(p) for p in seg) if seg else "")
            out = chunks + [fast(parts[-1], "explosively")]
        return out[:num_clips]

    def _load(self, name):
        return comfy.utils.load_torch_file(folder_paths.get_full_path("loras", name))

    @staticmethod
    def _parse_weights(text):
        try:
            nums = [float(x) for x in re.split(r'[,\s]+', text.strip()) if x != ""]
        except (ValueError, AttributeError):
            nums = []
        if not nums:
            return 1.0, 1.0, 1.0
        if len(nums) == 1:
            return nums[0], nums[0], nums[0]
        if len(nums) == 2:
            return nums[0], nums[1], nums[1]
        return nums[0], nums[1], nums[2]

    def _patch_clip(self, mb, mh, ml, slots, cache, inject):
        cb, ch, cl = mb, mh, ml
        used = []
        for (name, w_str) in slots:
            if name == "None":
                continue
            w_b, w_h, w_l = self._parse_weights(w_str)
            if w_b == 0.0 and w_h == 0.0 and w_l == 0.0:
                continue
            data = self._load(name)
            if w_b != 0.0:
                cb = apply_model_lora(cb, data, w_b)
            if w_h != 0.0:
                ch = apply_model_lora(ch, data, w_h)
            if w_l != 0.0:
                cl = apply_model_lora(cl, data, w_l)
            used.append(name)
        trig = []
        trig_dbg = []
        for n in dict.fromkeys(used):
            found = get_triggers(n, cache)
            trig.extend(found)
            trig_dbg.append(f"{os.path.basename(n)}:{len(found)}")
        self._last_trig_dbg = trig_dbg
        trig_str = ", ".join(dict.fromkeys(t for t in trig if t))
        prompt_trig = trig_str if inject else ""
        return cb, ch, cl, prompt_trig, trig_str

    def direct(self, image, model_base, model_high, model_low, essential_actions, num_clips,
               character_name, artistic_vibe, pacing, detail_level, inject_triggers,
               use_vision, openrouter_api_key, vlm_preset, vlm_model, llm_preset, llm_model,
               creativity, seed, **kw):

        n_clips = int(str(num_clips).split()[0])
        vlm = resolve_slug(VLM_PRESETS, vlm_preset, vlm_model, "qwen/qwen2.5-vl-72b-instruct")
        llm = resolve_slug(LLM_PRESETS, llm_preset, llm_model, "deepseek/deepseek-v3.2")
        user_name = "" if character_name.strip().upper() in ["AUTO", ""] else character_name.strip()
        key = resolve_api_key(openrouter_api_key)

        vision_caption = str(kw.get("vision_caption", "") or "").strip()
        anchors, dbg_v = "", ""
        if vision_caption:
            anchors = vision_caption
            dbg_v = "vision ok (local caption node)"
            print(f"πŸ‘ [Director] 둜컬 μΊ‘μ…˜ μž…λ ₯ μ‚¬μš© (JoyCaption λ“±):\n{anchors[:300]}")
        elif not use_vision:
            dbg_v = "vision OFF (ν† κΈ€ 확인)"
            print("ℹ️ [Director] use_vision=OFF - λΉ„μ „ κ±΄λ„ˆλœ€")
        elif not key:
            dbg_v = "no api key"
            print("❌ [Director] openrouter_api_key λΉ„μ–΄μžˆμŒ (μœ„μ ―λ„, OPENROUTER_API_KEY ν™˜κ²½λ³€μˆ˜λ„ μ—†μŒ) - λΉ„μ „/LLM λΆˆκ°€")
        elif image is None:
            dbg_v = "no image"
            print("❌ [Director] image μž…λ ₯ μ—†μŒ - IMAGE μ—°κ²° 확인")
        else:
            data_uri = tensor_to_base64(image)
            if data_uri:
                anchors, err = self._vision_analyze(data_uri, vlm, key, min(creativity, 0.5))
                dbg_v = err or "vision ok"
            else:
                dbg_v = "image encode failed / PIL missing"
        if not vision_caption and use_vision and key and image is not None and not anchors:
            print(f"⚠️ [Director] λΉ„μ „ 액컀가 λΉ„μ—ˆμŒ -> 이름/씬 κ·Όκ±° μ—†μŒ. 원인: {dbg_v}")

        if key:
            parsed, dbg_p = self._plan_shots(anchors, essential_actions, character_name,
                                             artistic_vibe, pacing, self._detail_hint(detail_level),
                                             llm, key, creativity, seed, num_clips=n_clips)
            if parsed:
                clips = [str(parsed[f"clip_{i}"]).strip() for i in range(1, n_clips + 1)]
                master = str(parsed.get("master_scene", "")).strip()
                cand = str(parsed.get("character", "")).strip()
                if not cand or cand.upper() == "AUTO":
                    cand = self._fallback_name(anchors)
                final_name = user_name or cand
            else:
                clips = self._local_split(essential_actions, num_clips=n_clips)
                master = anchors.replace("\n", " ")[:160]
                final_name = user_name or self._fallback_name(anchors)
        else:
            parsed, dbg_p = None, "no api key"
            clips = self._local_split(essential_actions, num_clips=n_clips)
            master = anchors.replace("\n", " ")[:160]
            final_name = user_name

        if len(clips) < 4:
            clips = clips + [clips[-1]] * (4 - len(clips))

        cache = _load_cache()
        manual_list = [t.strip() for t in kw.get("manual_triggers", "").split(',') if t.strip()]
        outs = []
        all_trigs_collected = []
        for c in range(1, 5):
            if c > n_clips:
                cb = ch = cl = ExecutionBlocker(None)
                prompt = self._build_final(clips[c - 1], master, final_name, "")
                outs.extend([cb, ch, cl, prompt])
                print(f"  [clip{c}] SKIPPED (num_clips={n_clips}) - ExecutionBlocker λ°˜ν™˜")
                continue
            slots = [(kw.get(f"c{c}_lora_{n}", "None"),
                      kw.get(f"c{c}_w_{n}", "1.0, 1.0, 1.0")) for n in range(1, 5)]
            cb, ch, cl, prompt_trig, always_trig = self._patch_clip(
                model_base, model_high, model_low, slots, cache, inject_triggers)
            all_trig = ", ".join(dict.fromkeys(
                ([prompt_trig] if prompt_trig else []) + manual_list)) \
                if (prompt_trig or manual_list) else ""
            if always_trig:
                all_trigs_collected.extend(t.strip() for t in always_trig.split(',') if t.strip())
            prompt = self._build_final(clips[c - 1], master, final_name, all_trig)
            outs.extend([cb, ch, cl, prompt])
            print(f"  [clip{c}] loras/triggers: {getattr(self, '_last_trig_dbg', [])} "
                  f"inject={inject_triggers}")

        triggers_out = ", ".join(dict.fromkeys(
            [t for t in all_trigs_collected if t] + manual_list))

        debug = f"vlm={dbg_v} | plan={dbg_p} | char={final_name} | clips={n_clips} ({n_clips*5}s)"
        print(f"\n🎬 [Dolphin Vision Director] {debug}\nANCHORS: {anchors[:120]}")
        for i in range(4):
            print(f"  CLIP {i+1}: {outs[i*4+3][:90]}")

        is_final = tuple(c == n_clips for c in range(1, 5))

        return tuple(outs) + (triggers_out, anchors, debug) + is_final


NODE_CLASS_MAPPINGS = {"DolphinVisionDirector": DolphinVisionDirector}
NODE_DISPLAY_NAME_MAPPINGS = {"DolphinVisionDirector": "πŸŽ¬πŸ‘ Dolphin Vision Director (Imageβ†’4Clips)"}