File size: 5,187 Bytes
77055e9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
import cv2
import numpy as np
import os
import sys
from pathlib import Path

# Add project root to path
sys.path.append(str(Path(__file__).parent.parent))

from src.utils.preprocess import GazePreprocessor

def create_visualization(image_path, output_path):
    print(f"Processing image: {image_path}")
    frame = cv2.imread(image_path)
    if frame is None:
        print(f"Error: Could not read image at {image_path}")
        return False
    
    h, w, _ = frame.shape
    # Ensure the image isn't too small or too large for the layout
    target_w = 1024
    scale = target_w / w
    frame_resized = cv2.resize(frame, (target_w, int(h * scale)))
    rh, rw, _ = frame_resized.shape
    
    preprocessor = GazePreprocessor(model_path='src/utils/face_landmarker.task')
    
    # 1. Get Landmarks
    landmarks = preprocessor.get_landmarks(frame)
    if landmarks is None:
        print("No face detected in the image.")
        return False

    # 2. Process Eyes (16x16 patches)
    # Use target_size=(64, 32) for normalization, then extract 16x16 patches
    left_eye_norm, _ = preprocessor.normalize_eye(frame, landmarks, 'left', target_size=(64, 32))
    right_eye_norm, _ = preprocessor.normalize_eye(frame, landmarks, 'right', target_size=(64, 32))
    
    # Extract 4 patches of 16x16 (total 32x32 area represented)
    # The user asked for "16x16", let's assume they want the 4 patches to be 16x16 each.
    left_patches = preprocessor.extract_patches(left_eye_norm, patch_size=16) 
    right_patches = preprocessor.extract_patches(right_eye_norm, patch_size=16)

    # 3. Draw Landmarks and Gaze on the main frame
    vis_frame = frame_resized.copy()
    
    # Draw Landmarks
    for lm in landmarks:
        x, y = int(lm.x * rw), int(lm.y * rh)
        cv2.circle(vis_frame, (x, y), 1, (0, 255, 0), -1)
    
    # Calculate Gaze Vector (Simulated for visualization)
    left_c_norm = np.mean([[landmarks[idx].x, landmarks[idx].y] for idx in preprocessor.LEFT_CORNERS], axis=0)
    right_c_norm = np.mean([[landmarks[idx].x, landmarks[idx].y] for idx in preprocessor.RIGHT_CORNERS], axis=0)
    
    lc = (int(left_c_norm[0] * rw), int(left_c_norm[1] * rh))
    rc = (int(right_c_norm[0] * rw), int(right_c_norm[1] * rh))
    
    # Yellow Gaze Arrows
    dx, dy = 80, -30
    cv2.arrowedLine(vis_frame, lc, (lc[0] + dx, lc[1] + dy), (0, 255, 255), 3, tipLength=0.3)
    cv2.arrowedLine(vis_frame, rc, (rc[0] + dx, rc[1] + dy), (0, 255, 255), 3, tipLength=0.3)

    # 4. Overlay Cropped Eyes at Corners
    # Create a small grid for the 4 patches
    def create_patch_grid(patches, size=16, display_size=120):
        # patches is (4, size, size)
        grid = np.zeros((size*2, size*2), dtype=np.uint8)
        grid[0:size, 0:size] = patches[0]
        grid[0:size, size:size*2] = patches[1]
        grid[size:size*2, 0:size] = patches[2]
        grid[size:size*2, size:size*2] = patches[3]
        
        # Upscale for visibility
        grid_colored = cv2.cvtColor(grid, cv2.COLOR_GRAY2BGR)
        grid_enlarged = cv2.resize(grid_colored, (display_size, display_size), interpolation=cv2.INTER_NEAREST)
        
        # Add border and label
        cv2.rectangle(grid_enlarged, (0, 0), (display_size-1, display_size-1), (255, 255, 255), 2)
        return grid_enlarged

    left_grid = create_patch_grid(left_patches, size=16, display_size=160)
    right_grid = create_patch_grid(right_patches, size=16, display_size=160)

    # Place in corners with some padding
    pad = 20
    # Top-Right for Right Eye patches
    vis_frame[pad:pad+160, rw-160-pad:rw-pad] = right_grid
    cv2.putText(vis_frame, "Right Eye (16x16 Patches)", (rw-160-pad, pad+160+20), 
                cv2.FONT_HERSHEY_SIMPLEX, 0.4, (255, 255, 255), 1)

    # Top-Left for Left Eye patches
    vis_frame[pad:pad+160, pad:pad+160] = left_grid
    cv2.putText(vis_frame, "Left Eye (16x16 Patches)", (pad, pad+160+20), 
                cv2.FONT_HERSHEY_SIMPLEX, 0.4, (255, 255, 255), 1)

    # 5. Add Infographic Title
    overlay = vis_frame.copy()
    cv2.rectangle(overlay, (0, rh-60), (rw, rh), (0, 0, 0), -1)
    cv2.addWeighted(overlay, 0.6, vis_frame, 0.4, 0, vis_frame)
    
    cv2.putText(vis_frame, "LIPE V2: Dual-State Pipeline | 16x16 Patch Embedder | Landmark-Guided Gaze", 
                (30, rh-25), cv2.FONT_HERSHEY_DUPLEX, 0.6, (255, 255, 255), 1)

    cv2.imwrite(output_path, vis_frame)
    print(f"New visualization saved to {output_path}")
    return True

if __name__ == "__main__":
    candidates = []
    # Try MPIIGaze images as they are standard
    mpii_base = Path("data/MPIIGaze/MPIIGaze/MPIIGaze/Data/Original/p00/day01")
    if mpii_base.exists():
        candidates += list(mpii_base.glob("000*.jpg"))[:10]
        
    candidates += list(Path("data/verification").glob("sample_*.png"))
    
    dest = "report/image/lipe_v2_16x16_viz.png"
    os.makedirs(os.path.dirname(dest), exist_ok=True)
    
    success = False
    for src in candidates:
        if src.exists():
            if create_visualization(str(src), dest):
                success = True
                break
    
    if not success:
        print("Error: Could not create visualization.")