File size: 5,821 Bytes
d176ecd
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
#!/usr/bin/env python3
"""Face identity helper: SCRFD landmarks + ArcFace (w600k_r50) embeddings via plain onnxruntime.

No `insightface` package needed (it would pull a CPU onnxruntime over onnxruntime-gpu); only the two ONNX files
from InsightFace's buffalo_l pack in models/. NOTE: those pretrained models are licensed for non-commercial
research use only.

Used by blur_video.py --keep to leave one person's face visible while blurring everyone else.
"""
from pathlib import Path

import cv2
import numpy as np
import onnxruntime as ort

MODELS = Path(__file__).resolve().parent / "models"
# ArcFace 112x112 alignment template (left eye, right eye, nose, left mouth, right mouth)
ARC_TEMPLATE = np.array([[38.2946, 51.6963], [73.5318, 51.5014], [56.0252, 71.7366],
                         [41.5493, 92.3655], [70.7299, 92.2041]], np.float32)


def _session(path, device):
    prov = ["CUDAExecutionProvider", "CPUExecutionProvider"] if device != "cpu" else ["CPUExecutionProvider"]
    so = ort.SessionOptions()
    so.log_severity_level = 3
    return ort.InferenceSession(str(path), sess_options=so, providers=prov)


def _nms(boxes, scores, thr=0.4):
    idx = cv2.dnn.NMSBoxes([[float(b[0]), float(b[1]), float(b[2] - b[0]), float(b[3] - b[1])] for b in boxes],
                           [float(s) for s in scores], 0.0, thr)
    return np.array(idx, dtype=int).reshape(-1)


class FaceID:
    def __init__(self, device="cpu"):
        self.det = _session(MODELS / "det_10g.onnx", device)
        self.rec = _session(MODELS / "w600k_r50.onnx", device)
        self.det_in = self.det.get_inputs()[0].name
        self.rec_in = self.rec.get_inputs()[0].name

    def detect(self, img, size=192, thr=0.4):
        """SCRFD on `img` letterboxed (top-left) into a size x size canvas.
        Returns boxes (N,4) xyxy, scores (N,), landmarks (N,5,2), all in `img` pixel coordinates."""
        h, w = img.shape[:2]
        s = size / max(h, w)
        nw, nh = max(1, int(round(w * s))), max(1, int(round(h * s)))
        canvas = np.zeros((size, size, 3), np.uint8)
        canvas[:nh, :nw] = cv2.resize(img, (nw, nh))
        blob = cv2.dnn.blobFromImage(canvas, 1 / 128.0, (size, size), (127.5, 127.5, 127.5), swapRB=True)
        outs = self.det.run(None, {self.det_in: blob})
        B, S, K = [], [], []
        for i, stride in enumerate((8, 16, 32)):
            sc, bb, kp = outs[i][:, 0], outs[i + 3] * stride, outs[i + 6] * stride
            g = size // stride
            ys, xs = np.mgrid[:g, :g]
            ctr = np.repeat(np.stack([xs, ys], -1).reshape(-1, 2) * stride, 2, axis=0).astype(np.float32)
            m = sc >= thr
            if not m.any():
                continue
            c, bb, kp = ctr[m], bb[m], kp[m]
            B.append(np.stack([c[:, 0] - bb[:, 0], c[:, 1] - bb[:, 1], c[:, 0] + bb[:, 2], c[:, 1] + bb[:, 3]], 1))
            K.append(np.stack([c[:, 0:1] + kp[:, 0::2], c[:, 1:2] + kp[:, 1::2]], -1))
            S.append(sc[m])
        if not B:
            return np.zeros((0, 4)), np.zeros(0), np.zeros((0, 5, 2))
        B, S, K = np.concatenate(B) / s, np.concatenate(S), np.concatenate(K) / s
        keep = _nms(B, S)
        return B[keep], S[keep], K[keep]

    def embed_aligned(self, img, lmk):
        M, _ = cv2.estimateAffinePartial2D(lmk.astype(np.float32), ARC_TEMPLATE, method=cv2.LMEDS)
        if M is None:
            return None
        face = cv2.warpAffine(img, M, (112, 112), borderValue=0)
        blob = cv2.dnn.blobFromImage(face, 1 / 127.5, (112, 112), (127.5, 127.5, 127.5), swapRB=True)
        e = self.rec.run(None, {self.rec_in: blob})[0][0]
        return e / (np.linalg.norm(e) + 1e-9)

    def embed_box(self, frame, box, expand=0.5, min_size=40):
        """Embedding of the face inside `box` (xyxy) of a full frame, or None when it is too small / not found.
        Crops the box expanded by `expand`, finds landmarks of the face nearest the box centre, aligns, embeds."""
        H, W = frame.shape[:2]
        x1, y1, x2, y2 = [float(v) for v in box[:4]]
        bw, bh = x2 - x1, y2 - y1
        if min(bw, bh) < min_size:
            return None
        cx1, cy1 = max(0, int(x1 - bw * expand)), max(0, int(y1 - bh * expand))
        cx2, cy2 = min(W, int(x2 + bw * expand)), min(H, int(y2 + bh * expand))
        crop = frame[cy1:cy2, cx1:cx2]
        if crop.size == 0:
            return None
        B, S, K = self.detect(crop)
        if len(B) == 0:
            return None
        tx, ty = (x1 + x2) / 2 - cx1, (y1 + y2) / 2 - cy1
        d = [np.hypot((b[0] + b[2]) / 2 - tx, (b[1] + b[3]) / 2 - ty) / max(bw, 1) for b in B]
        j = int(np.argmin(d))
        if d[j] > 0.35:  # the found face is not the one in `box`
            return None
        return self.embed_aligned(crop, K[j])

    def reference_embeddings(self, paths):
        """One embedding per reference photo (largest face in each). Raises if a photo has no usable face."""
        embs = []
        for p in paths:
            img = cv2.imread(str(p))
            if img is None:
                raise SystemExit(f"cannot read reference image {p}")
            B, S, K = self.detect(img, size=640, thr=0.5)
            if len(B) == 0:
                raise SystemExit(f"no face found in reference image {p}")
            if len(B) > 1:
                print(f"warning: {len(B)} faces in {p}; using the largest")
            j = int(np.argmax((B[:, 2] - B[:, 0]) * (B[:, 3] - B[:, 1])))
            e = self.embed_aligned(img, K[j])
            if e is None:
                raise SystemExit(f"could not align face in {p}")
            embs.append(e)
        return np.stack(embs)

    @staticmethod
    def similarity(emb, refs):
        """Best cosine similarity of `emb` against the reference embeddings."""
        return float((refs @ emb).max())