File size: 6,455 Bytes
536a29a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
"""Expression analysis: facial expression tags from pose + WD14 tags.

Extends the basic open_mouth/closed_eyes detection in wholebody_pose.py
with additional heuristics. Primary source is WD14 tag detection since
anime facial expression recognition via landmarks alone is unreliable.
"""
from __future__ import annotations

import numpy as np

from src.wholebody_pose import _vis


# WD14 tag → Danbooru canon mapping for expression tags
_WD_EXPRESSION_MAP = {
    "smile": "smile",
    "happy": "happy",
    "laughing": "laughing",
    "frown": "frown",
    "sad": "sad",
    "crying": "crying",
    "tears": "tears",
    "angry": "angry",
    "serious": "serious",
    "surprised": "surprised",
    "blush": "blush",
    "blushing": "blush",
    "red face": "blush",
    "wink": "wink",
    "confused": "confused",
    "expressionless": "expressionless",
    "poker face": "expressionless",
    "ahegao": "ahegao_face",
    "ahegao_face": "ahegao_face",
    # Note: open_mouth, closed_eyes, looking_*, wide_eyes are handled by pose
    # keypoints and should not be duplicated in expression tags
}


def _safe_dist(a: np.ndarray, b: np.ndarray) -> float:
    if not (_vis(a) and _vis(b)):
        return 0.0
    return float(np.sqrt((a[0] - b[0]) ** 2 + (a[1] - b[1]) ** 2))


def _vis_kpt(kp_arr: np.ndarray, idx: int, thresh: float = 0.15) -> bool:
    """Check if keypoint at index ``idx`` in ``kp_arr`` is visible."""
    if kp_arr.ndim != 2 or kp_arr.shape[0] <= idx:
        return False
    return kp_arr[idx, 2] >= thresh


def analyze_expression(face_kpts: np.ndarray,
                       body_kpts: np.ndarray,
                       wd14_tags: list[str]) -> list[str]:
    """Analyze facial expression from face landmarks + WD14 tags.

    Args:
        face_kpts: 68 face landmarks array (or full 133-kpt array).
        body_kpts: Body keypoints (for nose reference, first 17).
        wd14_tags: General tags from WD14 ensemble.

    Returns:
        List of Danbooru-style expression tags.
    """
    tags: set[str] = set()

    # --- Pass-through from WD14 (most reliable for anime expressions) ---
    for tag in wd14_tags:
        canon = _WD_EXPRESSION_MAP.get(tag)
        if canon:
            tags.add(canon)

    # --- Face landmark heuristics (supplement WD14) ---
    # Note: open_mouth, closed_eyes, and looking direction are also detected
    # by wholebody_pose._face_tags(). The expression analyzer supplements these
    # with WD14-derived tags (smile, blush, etc.) since landmark-based smile
    # detection is unreliable for anime art.
    if face_kpts is not None:
        # Normalize face array: if full 133-kpt array, extract face slice
        if face_kpts.shape[0] >= 91:
            face = face_kpts[23:91]  # 68 face landmarks
            body = face_kpts[:17]
        else:
            face = face_kpts[:68]
            body = body_kpts[:17] if body_kpts is not None and body_kpts.shape[0] >= 17 else None

        if face.shape[0] >= 68:
            # Mouth analysis
            if _check_open_mouth(face):
                tags.add("open_mouth")

            # Eye openness
            eye_tags = _check_eyes(face)
            tags.update(eye_tags)

            # Looking direction
            look_tags = _check_looking_direction(face, body)
            tags.update(look_tags)

    return sorted(tags) if tags else []


def _check_open_mouth(face: np.ndarray) -> bool:
    """Check if mouth is open using lip landmark distances."""
    # iBug 68-point face indices:
    # 48-54: outer lip contour (48 = left corner, 54 = right corner)
    # 55-59: inner lip upper
    # 60-64: inner lip lower
    # 65-67: mouth interior

    # Upper lip center (points 51-53) and lower lip center (points 57-59)
    ul = None
    ll = None

    if all(_vis_kpt(face, i, 0.15) for i in [51, 52, 53]):
        ul = np.mean(face[51:54], axis=0)
    if all(_vis_kpt(face, i, 0.15) for i in [57, 58, 59]):
        ll = np.mean(face[57:60], axis=0)

    if ul is not None and ll is not None:
        mouth_open_dist = _safe_dist(ul, ll)
        # Normalize by inter-eye distance
        leye_c = np.mean(face[36:42], axis=0) if all(_vis_kpt(face, i, 0.15) for i in range(36, 42)) else None
        reye_c = np.mean(face[42:48], axis=0) if all(_vis_kpt(face, i, 0.15) for i in range(42, 48)) else None
        if leye_c is not None and reye_c is not None:
            eye_dist = _safe_dist(leye_c, reye_c)
            if eye_dist > 5 and mouth_open_dist > 0.25 * eye_dist:
                return True
    return False


def _check_eyes(face: np.ndarray) -> set[str]:
    """Check eye openness using Eye Aspect Ratio (EAR)."""
    tags: set[str] = set()

    def _eye_ear(indices):
        if not all(_vis_kpt(face, i, 0.15) for i in indices):
            return None
        p = face[indices]
        v1 = _safe_dist(p[1], p[5])
        v2 = _safe_dist(p[2], p[4])
        h = _safe_dist(p[0], p[3])
        if h < 1e-3:
            return None
        return (v1 + v2) / (2 * h)

    # Left eye: 36-41, Right eye: 42-47
    l_ear = _eye_ear([36, 37, 38, 39, 40, 41])
    r_ear = _eye_ear([42, 43, 44, 45, 46, 47])

    if l_ear is not None and r_ear is not None:
        avg = (l_ear + r_ear) / 2
        if avg < 0.2:
            tags.add("closed_eyes")
        elif avg > 0.4:
            tags.add("wide_eyes")

    return tags


def _check_looking_direction(face: np.ndarray, body: np.ndarray) -> set[str]:
    """Detect looking direction from eye-to-nose offset."""
    tags: set[str] = set()

    if body is None or not _vis(body[0], 0.15):
        return tags

    nose_tip = body[0]
    leye_c = np.mean(face[36:42], axis=0) if all(_vis_kpt(face, i, 0.15) for i in range(36, 42)) else None
    reye_c = np.mean(face[42:48], axis=0) if all(_vis_kpt(face, i, 0.15) for i in range(42, 48)) else None

    if leye_c is not None and reye_c is not None:
        eye_mid_x = (leye_c[0] + reye_c[0]) / 2
        eye_mid_y = (leye_c[1] + reye_c[1]) / 2
        inter_eye = _safe_dist(leye_c, reye_c)
        if inter_eye > 5:
            nx = (nose_tip[0] - eye_mid_x) / inter_eye
            ny = (nose_tip[1] - eye_mid_y) / inter_eye

            if nx > 0.35:
                tags.add("looking_right")
            elif nx < -0.35:
                tags.add("looking_left")

            if ny > 0.35:
                tags.add("looking_up")
            elif ny < -0.35:
                tags.add("looking_down")

    return tags