File size: 6,510 Bytes
9bd3ee0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
"""
InsightFace face recognition provider (ArcFace via ONNX).

Uses the InsightFace `buffalo_s` model pack — the SMALL variant
(~50MB total) optimized for CPU inference.  Produces 512-d L2-normalized
embeddings; 99.7% accuracy on LFW.

Design:
  - Models are loaded LAZILY on first use via cores.onnx.get_session()
  - Models load exactly ONCE per process (cached)
  - Uses cores.face.cosine_similarity + best_match for gallery matching
  - If onnxruntime is not installed, is_available() returns False

Model files (auto-downloaded to data/models/):
  - det_500m.onnx  (~2MB)  — SCRFD face detector
  - w600k_mbf.onnx (~50MB) — ArcFace recognizer

License: MIT (InsightFace)
"""

from __future__ import annotations

from typing import Any

import numpy as np

from config.settings import Settings, settings as _default_settings
from cores.onnx import is_onnx_available, get_session, ensure_model
from cores.face import cosine_similarity, best_match
from cores.vision import to_rgb, BBox, crop_region
from pipeline.feature_extraction import PipelineOutput
from providers.base import BaseProvider, ProviderCapability


class InsightFaceProvider(BaseProvider):
    name = "insightface"
    capability = ProviderCapability.RECOGNITION

    # Model files in the buffalo_s pack
    DETECTOR_FILE = "det_500m.onnx"
    RECOGNIZER_FILE = "w600k_mbf.onnx"

    def __init__(self, settings: Settings | None = None) -> None:
        super().__init__(settings=settings or _default_settings)
        self._available = is_onnx_available()

    def is_available(self) -> bool:
        return self._available

    def _run(self, pipeline_output: PipelineOutput) -> tuple[dict, dict]:
        if not self._available:
            raise RuntimeError("onnxruntime not installed")

        # Lazy-load models (cached after first call)
        det_path = ensure_model(self.DETECTOR_FILE, settings=self._settings)
        rec_path = ensure_model(self.RECOGNIZER_FILE, settings=self._settings)
        det_session = get_session(det_path, self._settings)
        rec_session = get_session(rec_path, self._settings)

        img: np.ndarray = pipeline_output.image
        rgb = to_rgb(img)

        # Step 1: detect faces using the SCRFD detector
        detections = self._detect_faces(det_session, rgb)
        if not detections:
            raw = {"num_faces": 0, "matches": []}
            normalized = {"num_faces": 0, "matches": []}
            return raw, normalized

        # Step 2: for each detected face, compute embedding
        gallery = pipeline_output.gallery or {}
        matches: list[dict] = []
        embeddings: list[list[float]] = []

        for i, det in enumerate(detections):
            bbox = BBox(det["x"], det["y"], det["w"], det["h"])
            crop = crop_region(img, bbox, margin=0.15)
            embedding = self._embed(rec_session, crop)
            embeddings.append(embedding.tolist())

            # Match against gallery
            best_name, best_score, all_scores = best_match(
                embedding, gallery, metric="cosine",
            )
            threshold = self._settings.recognition_match_threshold
            matches.append({
                "query_face_index": i,
                "best_match": best_name if best_score >= threshold else None,
                "distance": 1.0 - best_score,
                "distances": all_scores,
                "box": det,
            })

        raw = {
            "num_faces": len(detections),
            "matches": matches,
            "embeddings_dim": 512,
            "model_pack": self._settings.insightface_model_pack,
        }
        normalized = {
            "num_faces": len(detections),
            "matches": matches,
            "embeddings": embeddings,
        }
        return raw, normalized

    # ------------------------------------------------------------------ #
    # Face detection (SCRFD) — minimal postprocessing
    # ------------------------------------------------------------------ #
    def _detect_faces(self, session, rgb: np.ndarray) -> list[dict]:
        """Run SCRFD detection.  Returns list of {x, y, w, h, confidence}."""
        h, w = rgb.shape[:2]
        # SCRFD expects 640x640 input
        import cv2
        input_h, input_w = 640, 640
        scale = min(input_h / h, input_w / w)
        new_h, new_w = int(h * scale), int(w * scale)
        resized = cv2.resize(rgb, (new_w, new_h))
        padded = np.zeros((input_h, input_w, 3), dtype=np.float32)
        padded[:new_h, :new_w] = resized.astype(np.float32)
        # Normalize
        padded = (padded - 127.5) / 128.0
        padded = padded.transpose(2, 0, 1)[None]  # NCHW

        outputs = session.run(padded)
        # SCRFD output format varies; take the first output as scores
        # and the second as boxes.  This is a simplified parser.
        scores = outputs[0]  # (1, N)
        boxes = outputs[1] if len(outputs) > 1 else None
        if boxes is None:
            return []

        # Filter by confidence
        threshold = 0.5
        detections: list[dict] = []
        for i in range(min(len(scores), len(boxes))):
            if scores[i] < threshold:
                continue
            box = boxes[i]
            # box is [x1, y1, x2, y2] in input scale
            x1 = int(box[0] / scale)
            y1 = int(box[1] / scale)
            x2 = int(box[2] / scale)
            y2 = int(box[3] / scale)
            x1, y1 = max(0, x1), max(0, y1)
            x2, y2 = min(w, x2), min(h, y2)
            detections.append({
                "x": x1, "y": y1, "w": x2 - x1, "h": y2 - y1,
                "confidence": float(scores[i]),
            })
        return detections

    # ------------------------------------------------------------------ #
    # Embedding generation (ArcFace)
    # ------------------------------------------------------------------ #
    def _embed(self, session, crop: np.ndarray) -> np.ndarray:
        """Generate 512-d embedding from a face crop."""
        import cv2
        if crop.size == 0:
            return np.zeros(512, dtype=np.float32)
        rgb = to_rgb(crop)
        resized = cv2.resize(rgb, (112, 112))
        normalized = (resized.astype(np.float32) - 127.5) / 128.0
        nchw = normalized.transpose(2, 0, 1)[None]
        output = session.run_single(nchw)
        embedding = output[0]
        # L2 normalize
        norm = np.linalg.norm(embedding)
        if norm > 0:
            embedding = embedding / norm
        return embedding