visual-search-api2 / src /models.py
AdarshDRC's picture
fix : face not detecting
4d03437
Raw
History Blame
15.6 kB
# src/models.py — Enterprise Lens V3
# ════════════════════════════════════════════════════════════════════
# Face Lane : InsightFace (YuNet detection + ArcFace 512-D encoding)
# • Replaces DeepFace + RetinaFace + GhostFaceNet entirely
# • 3-5x faster on CPU, handles small faces (≥20×20 px)
# • Stores one 512-D vector PER face (not per image)
# • Each vector carries a base64 face-crop thumbnail
# Object Lane: SigLIP + DINOv2 fused 1536-D (unchanged from V2)
# ════════════════════════════════════════════════════════════════════
import os
os.environ["TF_CPP_MIN_LOG_LEVEL"] = "3"
import asyncio
import base64
import functools
import hashlib
import io
import cv2
import numpy as np
import threading
import torch
import torch.nn.functional as F
from PIL import Image
from transformers import AutoImageProcessor, AutoModel, AutoProcessor
from ultralytics import YOLO
# ── InsightFace ───────────────────────────────────────────────────
try:
import insightface
from insightface.app import FaceAnalysis
INSIGHTFACE_AVAILABLE = True
except ImportError:
INSIGHTFACE_AVAILABLE = False
print("⚠️ insightface not installed — face lane disabled")
# ── Constants ─────────────────────────────────────────────────────
YOLO_PERSON_CLASS_ID = 0
MIN_FACE_SIZE = 20 # minimum face width/height in pixels
MAX_FACES_PER_IMAGE = 10 # cap faces per image for upload
MAX_CROPS = 6 # max YOLO object crops per image
MAX_IMAGE_SIZE = 640 # resize longest edge before inference (V3: 640 vs V2: 512)
FACE_CROP_THUMB_SIZE = 112 # face thumbnail size stored in Pinecone metadata
FACE_CROP_QUALITY = 75 # JPEG quality for stored thumbnails
def _resize_pil(img: Image.Image, max_side: int = MAX_IMAGE_SIZE) -> Image.Image:
w, h = img.size
if max(w, h) <= max_side:
return img
scale = max_side / max(w, h)
return img.resize((int(w * scale), int(h * scale)), Image.LANCZOS)
def _img_hash(image_path: str) -> str:
h = hashlib.md5()
with open(image_path, "rb") as f:
h.update(f.read(65536))
return h.hexdigest()
def _crop_to_b64(img_np: np.ndarray, bbox: list, thumb_size: int = FACE_CROP_THUMB_SIZE) -> str:
"""Crop face from image, resize to thumbnail, return as base64 JPEG string."""
x, y, w, h = bbox
x, y = max(0, x), max(0, y)
# Add 20% padding for more natural face crop
pad_x = int(w * 0.2)
pad_y = int(h * 0.2)
x1 = max(0, x - pad_x)
y1 = max(0, y - pad_y)
x2 = min(img_np.shape[1], x + w + pad_x)
y2 = min(img_np.shape[0], y + h + pad_y)
face_crop = img_np[y1:y2, x1:x2]
if face_crop.size == 0:
return ""
# Resize to thumbnail
face_pil = Image.fromarray(face_crop[..., ::-1]) # BGR → RGB
face_pil = face_pil.resize((thumb_size, thumb_size), Image.LANCZOS)
buf = io.BytesIO()
face_pil.save(buf, format="JPEG", quality=FACE_CROP_QUALITY)
return base64.b64encode(buf.getvalue()).decode()
class AIModelManager:
def __init__(self):
self.device = (
"cuda" if torch.cuda.is_available()
else ("mps" if torch.backends.mps.is_available() else "cpu")
)
print(f"Loading models onto: {self.device.upper()}...")
# ── Object Lane: SigLIP + DINOv2 (unchanged) ─────────────
self.siglip_processor = AutoProcessor.from_pretrained(
"google/siglip-base-patch16-224", use_fast=True)
self.siglip_model = AutoModel.from_pretrained(
"google/siglip-base-patch16-224").to(self.device).eval()
self.dinov2_processor = AutoImageProcessor.from_pretrained("facebook/dinov2-base")
self.dinov2_model = AutoModel.from_pretrained(
"facebook/dinov2-base").to(self.device).eval()
if self.device == "cuda":
self.siglip_model = self.siglip_model.half()
self.dinov2_model = self.dinov2_model.half()
# ── YOLO for object segmentation ─────────────────────────
self.yolo = YOLO("yolo11n-seg.pt")
# ── Face Lane: InsightFace (YuNet + ArcFace) ─────────────
self.face_app = None
print(f"🔍 INSIGHTFACE_AVAILABLE = {INSIGHTFACE_AVAILABLE}")
if INSIGHTFACE_AVAILABLE:
try:
import insightface
print(f"🔍 InsightFace version: {insightface.__version__}")
model_name = "buffalo_l" if self.device == "cuda" else "buffalo_sc"
print(f"🔍 Loading InsightFace model: {model_name}")
self.face_app = FaceAnalysis(name=model_name)
self.face_app.prepare(
ctx_id=0 if self.device == "cuda" else -1,
det_size=(640, 640),
)
# Test with a blank image to confirm models loaded
import numpy as _np
test_img = _np.zeros((112, 112, 3), dtype=_np.uint8)
_ = self.face_app.get(test_img)
print(f"✅ InsightFace ({model_name}) loaded — ArcFace face lane ACTIVE")
except Exception as e:
import traceback
print(f"❌ InsightFace init FAILED: {e}")
print(traceback.format_exc())
self.face_app = None
else:
print("❌ InsightFace NOT installed — run: pip install insightface onnxruntime")
self._cache = {}
self._cache_maxsize = 128
# InsightFace ONNX runtime is NOT thread-safe
# This lock ensures only one inference runs at a time
self._face_lock = threading.Lock()
print("✅ Models ready!")
# ── Object Lane batched embedding ────────────────────────────
def _embed_crops_batch(self, crops: list) -> list:
if not crops:
return []
with torch.no_grad():
sig_in = self.siglip_processor(images=crops, return_tensors="pt", padding=True)
sig_in = {k: v.to(self.device) for k, v in sig_in.items()}
if self.device == "cuda":
sig_in = {k: v.half() if v.dtype == torch.float32 else v
for k, v in sig_in.items()}
sig_out = self.siglip_model.get_image_features(**sig_in)
if hasattr(sig_out, "image_embeds"): sig_out = sig_out.image_embeds
elif isinstance(sig_out, tuple): sig_out = sig_out[0]
sig_vecs = F.normalize(sig_out.float(), p=2, dim=1).cpu()
dino_in = self.dinov2_processor(images=crops, return_tensors="pt")
dino_in = {k: v.to(self.device) for k, v in dino_in.items()}
if self.device == "cuda":
dino_in = {k: v.half() if v.dtype == torch.float32 else v
for k, v in dino_in.items()}
dino_out = self.dinov2_model(**dino_in)
dino_vecs = F.normalize(
dino_out.last_hidden_state[:, 0, :].float(), p=2, dim=1).cpu()
fused = F.normalize(torch.cat([sig_vecs, dino_vecs], dim=1), p=2, dim=1)
return [fused[i].numpy() for i in range(len(crops))]
# ── V3 Face detection + encoding ─────────────────────────────
def _detect_and_encode_faces(self, img_np: np.ndarray) -> list:
"""
Detect ALL faces in image using InsightFace (YuNet + ArcFace).
Returns list of dicts:
{
"type": "face",
"vector": np.ndarray (512-D ArcFace embedding),
"face_idx": int,
"bbox": [x, y, w, h],
"face_crop": str (base64 JPEG thumbnail),
"det_score": float (detection confidence)
}
"""
if self.face_app is None:
print("⚠️ face_app is None — InsightFace not loaded!")
return []
try:
print(f"🔍 Running InsightFace on image shape: {img_np.shape}")
# InsightFace expects BGR numpy array
if img_np.shape[2] == 3 and img_np.dtype == np.uint8:
bgr = img_np[..., ::-1].copy() # RGB → BGR
else:
bgr = img_np.copy()
with self._face_lock:
faces = self.face_app.get(bgr)
print(f"🔍 InsightFace raw detection: {len(faces)} faces found")
results = []
for idx, face in enumerate(faces):
if idx >= MAX_FACES_PER_IMAGE:
break
# Get bounding box
bbox = face.bbox.astype(int) # [x1, y1, x2, y2]
x1, y1, x2, y2 = bbox
w, h = x2 - x1, y2 - y1
# Skip tiny faces
if w < MIN_FACE_SIZE or h < MIN_FACE_SIZE:
continue
# Get ArcFace embedding (already L2-normalised by InsightFace)
if face.embedding is None:
continue
vec = face.embedding.astype(np.float32)
# Re-normalise just to be safe
norm = np.linalg.norm(vec)
if norm > 0:
vec = vec / norm
# Generate face crop thumbnail for UI
face_crop_b64 = _crop_to_b64(
bgr, [x1, y1, w, h], FACE_CROP_THUMB_SIZE)
results.append({
"type": "face",
"vector": vec,
"face_idx": idx,
"bbox": [int(x1), int(y1), int(w), int(h)],
"face_crop": face_crop_b64,
"det_score": float(face.det_score) if hasattr(face, "det_score") else 1.0,
})
print(f"👤 Detected {len(results)} face(s) via InsightFace ArcFace")
return results
except Exception as e:
print(f"🟠 InsightFace error: {e} — falling back to object lane")
return []
# ── Main process_image ────────────────────────────────────────
def process_image(
self,
image_path: str,
is_query: bool = False,
detect_faces: bool = True,
) -> list:
"""
Returns list of vector dicts for upload or search.
Upload mode (is_query=False):
- Face vectors include bbox + face_crop for Pinecone metadata
- Object vectors include full-image + YOLO crops
Query mode (is_query=True):
- Same structure — main.py handles grouping for search response
"""
cache_key = f"{_img_hash(image_path)}_{detect_faces}_{is_query}"
if cache_key in self._cache:
print("⚡ Cache hit — skipping inference")
return self._cache[cache_key]
extracted = []
original_pil = Image.open(image_path).convert("RGB")
img_np = np.array(original_pil) # RGB, uint8
faces_found = False
# ── FACE LANE ────────────────────────────────────────────
if detect_faces:
# Resize for face detection (640px for small face detection)
detect_pil = _resize_pil(original_pil, 640)
detect_np = np.array(detect_pil)
face_results = self._detect_and_encode_faces(detect_np)
if face_results:
faces_found = True
# Scale bbox back to original image size if resized
scale_x = original_pil.width / detect_pil.width
scale_y = original_pil.height / detect_pil.height
for fr in face_results:
if scale_x != 1.0 or scale_y != 1.0:
bx, by, bw, bh = fr["bbox"]
fr["bbox"] = [
int(bx * scale_x), int(by * scale_y),
int(bw * scale_x), int(bh * scale_y),
]
extracted.append(fr)
# ── OBJECT LANE ──────────────────────────────────────────
# Always run object lane — even if faces found
# (image may contain both people and objects)
crops_pil = [_resize_pil(original_pil, MAX_IMAGE_SIZE)] # full-image always
yolo_results = self.yolo(image_path, conf=0.5, verbose=False)
for r in yolo_results:
if r.masks is not None:
for seg_idx, mask_xy in enumerate(r.masks.xy):
cls_id = int(r.boxes.cls[seg_idx].item())
# Skip person crops if face lane already handled them
if faces_found and cls_id == YOLO_PERSON_CLASS_ID:
print("🔵 PERSON crop skipped — face lane active")
continue
polygon = np.array(mask_xy, dtype=np.int32)
if len(polygon) < 3:
continue
x, y, w, h = cv2.boundingRect(polygon)
if w < 30 or h < 30:
continue
crop = original_pil.crop((x, y, x + w, y + h))
crops_pil.append(crop)
if len(crops_pil) >= MAX_CROPS + 1:
break
elif r.boxes is not None:
for box in r.boxes:
cls_id = int(box.cls.item())
if faces_found and cls_id == YOLO_PERSON_CLASS_ID:
continue
x1, y1, x2, y2 = box.xyxy[0].tolist()
if (x2 - x1) < 30 or (y2 - y1) < 30:
continue
crop = original_pil.crop((x1, y1, x2, y2))
crops_pil.append(crop)
if len(crops_pil) >= MAX_CROPS + 1:
break
crops = [_resize_pil(c, MAX_IMAGE_SIZE) for c in crops_pil]
print(f"🧠 Embedding {len(crops)} object crop(s) in one batch …")
obj_vecs = self._embed_crops_batch(crops)
for vec in obj_vecs:
extracted.append({"type": "object", "vector": vec})
# Cache result
if len(self._cache) >= self._cache_maxsize:
oldest = next(iter(self._cache))
del self._cache[oldest]
self._cache[cache_key] = extracted
return extracted
async def process_image_async(
self,
image_path: str,
is_query: bool = False,
detect_faces: bool = True,
) -> list:
loop = asyncio.get_event_loop()
return await loop.run_in_executor(
None,
functools.partial(self.process_image, image_path, is_query, detect_faces),
)