File size: 6,510 Bytes
9bd3ee0 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 | """
InsightFace face recognition provider (ArcFace via ONNX).
Uses the InsightFace `buffalo_s` model pack — the SMALL variant
(~50MB total) optimized for CPU inference. Produces 512-d L2-normalized
embeddings; 99.7% accuracy on LFW.
Design:
- Models are loaded LAZILY on first use via cores.onnx.get_session()
- Models load exactly ONCE per process (cached)
- Uses cores.face.cosine_similarity + best_match for gallery matching
- If onnxruntime is not installed, is_available() returns False
Model files (auto-downloaded to data/models/):
- det_500m.onnx (~2MB) — SCRFD face detector
- w600k_mbf.onnx (~50MB) — ArcFace recognizer
License: MIT (InsightFace)
"""
from __future__ import annotations
from typing import Any
import numpy as np
from config.settings import Settings, settings as _default_settings
from cores.onnx import is_onnx_available, get_session, ensure_model
from cores.face import cosine_similarity, best_match
from cores.vision import to_rgb, BBox, crop_region
from pipeline.feature_extraction import PipelineOutput
from providers.base import BaseProvider, ProviderCapability
class InsightFaceProvider(BaseProvider):
name = "insightface"
capability = ProviderCapability.RECOGNITION
# Model files in the buffalo_s pack
DETECTOR_FILE = "det_500m.onnx"
RECOGNIZER_FILE = "w600k_mbf.onnx"
def __init__(self, settings: Settings | None = None) -> None:
super().__init__(settings=settings or _default_settings)
self._available = is_onnx_available()
def is_available(self) -> bool:
return self._available
def _run(self, pipeline_output: PipelineOutput) -> tuple[dict, dict]:
if not self._available:
raise RuntimeError("onnxruntime not installed")
# Lazy-load models (cached after first call)
det_path = ensure_model(self.DETECTOR_FILE, settings=self._settings)
rec_path = ensure_model(self.RECOGNIZER_FILE, settings=self._settings)
det_session = get_session(det_path, self._settings)
rec_session = get_session(rec_path, self._settings)
img: np.ndarray = pipeline_output.image
rgb = to_rgb(img)
# Step 1: detect faces using the SCRFD detector
detections = self._detect_faces(det_session, rgb)
if not detections:
raw = {"num_faces": 0, "matches": []}
normalized = {"num_faces": 0, "matches": []}
return raw, normalized
# Step 2: for each detected face, compute embedding
gallery = pipeline_output.gallery or {}
matches: list[dict] = []
embeddings: list[list[float]] = []
for i, det in enumerate(detections):
bbox = BBox(det["x"], det["y"], det["w"], det["h"])
crop = crop_region(img, bbox, margin=0.15)
embedding = self._embed(rec_session, crop)
embeddings.append(embedding.tolist())
# Match against gallery
best_name, best_score, all_scores = best_match(
embedding, gallery, metric="cosine",
)
threshold = self._settings.recognition_match_threshold
matches.append({
"query_face_index": i,
"best_match": best_name if best_score >= threshold else None,
"distance": 1.0 - best_score,
"distances": all_scores,
"box": det,
})
raw = {
"num_faces": len(detections),
"matches": matches,
"embeddings_dim": 512,
"model_pack": self._settings.insightface_model_pack,
}
normalized = {
"num_faces": len(detections),
"matches": matches,
"embeddings": embeddings,
}
return raw, normalized
# ------------------------------------------------------------------ #
# Face detection (SCRFD) — minimal postprocessing
# ------------------------------------------------------------------ #
def _detect_faces(self, session, rgb: np.ndarray) -> list[dict]:
"""Run SCRFD detection. Returns list of {x, y, w, h, confidence}."""
h, w = rgb.shape[:2]
# SCRFD expects 640x640 input
import cv2
input_h, input_w = 640, 640
scale = min(input_h / h, input_w / w)
new_h, new_w = int(h * scale), int(w * scale)
resized = cv2.resize(rgb, (new_w, new_h))
padded = np.zeros((input_h, input_w, 3), dtype=np.float32)
padded[:new_h, :new_w] = resized.astype(np.float32)
# Normalize
padded = (padded - 127.5) / 128.0
padded = padded.transpose(2, 0, 1)[None] # NCHW
outputs = session.run(padded)
# SCRFD output format varies; take the first output as scores
# and the second as boxes. This is a simplified parser.
scores = outputs[0] # (1, N)
boxes = outputs[1] if len(outputs) > 1 else None
if boxes is None:
return []
# Filter by confidence
threshold = 0.5
detections: list[dict] = []
for i in range(min(len(scores), len(boxes))):
if scores[i] < threshold:
continue
box = boxes[i]
# box is [x1, y1, x2, y2] in input scale
x1 = int(box[0] / scale)
y1 = int(box[1] / scale)
x2 = int(box[2] / scale)
y2 = int(box[3] / scale)
x1, y1 = max(0, x1), max(0, y1)
x2, y2 = min(w, x2), min(h, y2)
detections.append({
"x": x1, "y": y1, "w": x2 - x1, "h": y2 - y1,
"confidence": float(scores[i]),
})
return detections
# ------------------------------------------------------------------ #
# Embedding generation (ArcFace)
# ------------------------------------------------------------------ #
def _embed(self, session, crop: np.ndarray) -> np.ndarray:
"""Generate 512-d embedding from a face crop."""
import cv2
if crop.size == 0:
return np.zeros(512, dtype=np.float32)
rgb = to_rgb(crop)
resized = cv2.resize(rgb, (112, 112))
normalized = (resized.astype(np.float32) - 127.5) / 128.0
nchw = normalized.transpose(2, 0, 1)[None]
output = session.run_single(nchw)
embedding = output[0]
# L2 normalize
norm = np.linalg.norm(embedding)
if norm > 0:
embedding = embedding / norm
return embedding
|